diff --git a/docs/customizer/manage-customization-jobs/get-job-status.mdx b/docs/customizer/manage-customization-jobs/get-job-status.mdx index 21ffd6d0fc..0ca20c7e42 100644 --- a/docs/customizer/manage-customization-jobs/get-job-status.mdx +++ b/docs/customizer/manage-customization-jobs/get-job-status.mdx @@ -9,7 +9,7 @@ Get detailed execution status for a customization job, including step-by-step pr This endpoint provides granular execution details including: -- **Step-level status**: `model-and-dataset-download` → `customization-training-job` → `model-upload` → `model-entity-creation` +- **Step-level status**: `model-and-dataset-download` → `training` → `model-upload` → `model-entity-creation` - **Training metrics**: `step`, `epoch`, `loss`, `lr` (learning rate), `grad_norm`, `val_loss` - **Progress tracking**: `downloaded_files`, `uploaded_bytes`, `progress_pct` @@ -45,7 +45,7 @@ client = NeMoPlatform( workspace="default", ) -# Get job status (use the job name returned by jobs.create) +# Get job status (use the job name returned at submission) job_name = "automodel-a1b2c3d4e5f6" status = client.jobs.get_status(name=job_name, workspace="default") @@ -55,7 +55,7 @@ print(f"Status: {status.status}") # Check step-level status and training progress for step in status.steps or []: print(f" Step '{step.name}': {step.status}") - if step.name == "customization-training-job": + if step.name == "training": for task in step.tasks or []: task_details = task.status_details or {} current_step = task_details.get("step") @@ -64,6 +64,21 @@ for step in status.steps or []: print(f" Progress: {current_step}/{max_steps}") ``` +### CLI + +```bash +nemo jobs get-status automodel-a1b2c3d4e5f6 --workspace default +``` + +### REST API + +```bash +curl -X GET \ + "${NMP_BASE_URL}/apis/jobs/v2/workspaces/default/jobs/automodel-a1b2c3d4e5f6/status" \ + -H 'Accept: application/json' \ + | jq +``` + **Active Job (Training in Progress)** @@ -111,7 +126,7 @@ for step in status.steps or []: { "id": "platform-job-step-9kme8ibxDGES4t9TZvLp4X", "error_details": {}, - "name": "customization-training-job", + "name": "training", "status": "active", "status_details": { "message": "Job is running" @@ -187,7 +202,7 @@ for step in status.steps or []: { "id": "platform-job-step-9kme8ibxDGES4t9TZvLp4X", "error_details": {}, - "name": "customization-training-job", + "name": "training", "status": "completed", "status_details": { "message": "Job completed successfully with exit code 0" diff --git a/docs/customizer/tutorials/distillation-customization-job.ipynb b/docs/customizer/tutorials/distillation-customization-job.ipynb index e095144ea3..00d6515395 100644 --- a/docs/customizer/tutorials/distillation-customization-job.ipynb +++ b/docs/customizer/tutorials/distillation-customization-job.ipynb @@ -1,927 +1,930 @@ { - "cells": [ - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "\n", - "\n", - "\n", - "# Knowledge Distillation Customization\n", - "\n", - "Learn how to train a smaller student model to mimic a larger teacher model using knowledge distillation (KD).\n", - "\n", - "## About\n", - "\n", - "Knowledge Distillation transfers knowledge from a large **teacher** model to a smaller **student** model. During training, the student learns to match the teacher's output probability distribution, producing a compact model that retains much of the teacher's capability.\n", - "\n", - "**What you can achieve with KD:**\n", - "\n", - "- **Compress models:** Distill a 3B model into a 1B model for faster inference and lower deployment costs\n", - "- **Reduce latency:** Deploy a smaller model that responds faster while preserving quality\n", - "- **Lower resource requirements:** Serve a distilled model on fewer GPUs\n", - "\n", - "### KD vs SFT: Understanding the Trade-offs\n", - "\n", - "| Aspect | Full SFT | Knowledge Distillation |\n", - "| --- | --- | --- |\n", - "| **Training signal** | Ground-truth labels only | Teacher's soft probability distribution + labels |\n", - "| **Knowledge source** | Dataset examples | Teacher model's learned representations |\n", - "| **Output model size** | Same as input model | Typically a smaller student model |\n", - "| **GPU requirements** | Needs to fit one model | Needs to fit both teacher and student in memory |\n", - "| **Best for** | Domain adaptation, new knowledge injection | Model compression, latency reduction |\n", - "\n", - "### Key Parameters\n", - "\n", - "| Parameter | Default | Description |\n", - "| --- | --- | --- |\n", - "| `teacher_model` | *(required)* | Teacher model entity URN (e.g., `default/llama-3-2-3b-teacher`) |\n", - "| `teacher_precision` | `bf16` | Precision for the frozen teacher (`bf16`, `fp16`, `fp32`). Lower = less memory |\n", - "| `distillation_ratio` | `0.5` | Balance between CE loss and KD loss. `0.0` = CE only, `1.0` = KD only |\n", - "| `distillation_temperature` | `1.0` | Softmax temperature. Higher = softer distributions, more knowledge transfer |\n", - "\n", - "### Workflow Overview\n", - "\n", - "This tutorial follows a complete distillation pipeline:\n", - "\n", - "1. **Fine-tune the teacher** (SFT on the task dataset) so it learns the domain\n", - "2. **Establish a baseline** by deploying the base student model and measuring ROUGE scores\n", - "3. **Distill into the student** using the fine-tuned teacher's soft targets\n", - "4. **Evaluate the distilled student** and compare ROUGE scores against the baseline\n", - "\n", - "**When to choose KD:**\n", - "\n", - "- You have a high-quality large model and want a smaller, faster version\n", - "- Deployment latency or cost is a constraint\n", - "- The teacher and student share the same vocabulary (e.g., both are Llama models)\n", - "\n", - "**When to choose SFT instead:** Refer to the [Full SFT tutorial](./sft-customization-job) when you want to train a model directly on labeled data without a teacher." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Prerequisites\n", - "\n", - "Before starting this tutorial, ensure you have:\n", - "\n", - "1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n", - "2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n", - "3. **Installed evaluation dependencies:**\n", - "\n", - "```sh\n", - "pip install evaluate rouge_score datasets\n", - "```" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Quick Start\n", - "\n", - "### 1. Initialize SDK\n", - "\n", - "The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](../../get-started/quickstart.md) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable:\n", - "\n", - "```sh\n", - "export NMP_BASE_URL=\n", - "```" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import json\n", - "import os\n", - "import time\n", - "import uuid\n", - "from pathlib import Path\n", - "\n", - "from nemo_platform import NeMoPlatform, ConflictError\n", - "\n", - "NMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\n", - "client = NeMoPlatform(\n", - " base_url=NMP_BASE_URL,\n", - " workspace=\"default\"\n", - ")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 2. Prepare Dataset\n", - "\n", - "Knowledge distillation uses the same dataset formats as SFT. We use the SQuAD dataset for both teacher training and distillation so that the teacher first learns the task, then transfers that knowledge to the student.\n", - "\n", - "We also hold out a small **test split** for ROUGE evaluation at the end." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from datasets import load_dataset, DatasetDict\n", - "\n", - "print(\"Loading dataset rajpurkar/squad\")\n", - "raw_dataset = load_dataset(\"rajpurkar/squad\")\n", - "if not isinstance(raw_dataset, DatasetDict):\n", - " raise ValueError(\"Dataset does not contain expected splits\")\n", - "\n", - "print(\"Loaded dataset\")\n", - "\n", - "SEED = 1234\n", - "TRAINING_SIZE = 3000\n", - "VALIDATION_SIZE = 300\n", - "TEST_SIZE = 100\n", - "DATASET_PATH = Path(\"kd-dataset\").absolute()\n", - "\n", - "os.makedirs(DATASET_PATH, exist_ok=True)\n", - "\n", - "train_set = raw_dataset.get('train')\n", - "split = train_set.train_test_split(test_size=0.05, seed=SEED)\n", - "\n", - "train_ds = split['train'].select(range(min(TRAINING_SIZE, len(split['train']))))\n", - "val_ds = split['test'].select(range(min(VALIDATION_SIZE, len(split['test']))))\n", - "test_ds = split['test'].select(range(VALIDATION_SIZE, min(VALIDATION_SIZE + TEST_SIZE, len(split['test']))))\n", - "\n", - "\n", - "def convert_squad(example):\n", - " \"\"\"Convert SQuAD format to prompt/completion format.\"\"\"\n", - " prompt = f\"Context: {example['context']} Question: {example['question']} Answer:\"\n", - " completion = example[\"answers\"][\"text\"][0]\n", - " return {\"prompt\": prompt, \"completion\": completion}\n", - "\n", - "\n", - "def write_jsonl(dataset, path):\n", - " with open(path, \"w\", encoding=\"utf-8\") as f:\n", - " for example in dataset:\n", - " f.write(json.dumps(convert_squad(example)) + \"\\n\")\n", - "\n", - "\n", - "def write_test_jsonl(dataset, path):\n", - " \"\"\"Save test split with raw context/question for chat-style evaluation.\"\"\"\n", - " with open(path, \"w\", encoding=\"utf-8\") as f:\n", - " for example in dataset:\n", - " f.write(json.dumps({\n", - " \"context\": example[\"context\"],\n", - " \"question\": example[\"question\"],\n", - " \"completion\": example[\"answers\"][\"text\"][0],\n", - " }) + \"\\n\")\n", - "\n", - "\n", - "write_jsonl(train_ds, f\"{DATASET_PATH}/training.jsonl\")\n", - "write_jsonl(val_ds, f\"{DATASET_PATH}/validation.jsonl\")\n", - "write_test_jsonl(test_ds, f\"{DATASET_PATH}/testing.jsonl\")\n", - "\n", - "print(f\"Training: {len(train_ds)} rows\")\n", - "print(f\"Validation: {len(val_ds)} rows\")\n", - "print(f\"Test: {len(test_ds)} rows\")\n", - "\n", - "with open(f\"{DATASET_PATH}/training.jsonl\", 'r') as f:\n", - " sample = json.loads(f.readline())\n", - " print(f\"\\nSample prompt: {sample['prompt'][:150]}...\")\n", - " print(f\"Sample completion: {sample['completion']}\")" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "DATASET_NAME = \"kd-dataset\"\n", - "\n", - "try:\n", - " client.files.filesets.create(\n", - " workspace=\"default\",\n", - " name=DATASET_NAME,\n", - " description=\"Knowledge distillation training data\"\n", - " )\n", - " print(f\"Created fileset: {DATASET_NAME}\")\n", - "except ConflictError:\n", - " print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n", - "\n", - "client.files.upload(\n", - " local_path=DATASET_PATH,\n", - " remote_path=\"\",\n", - " fileset=DATASET_NAME,\n", - " workspace=\"default\"\n", - ")\n", - "\n", - "print(\"Uploaded files:\")\n", - "print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 3. Secrets Setup\n", - "\n", - "In this tutorial we use two Llama 3.2 Instruct models from HuggingFace:\n", - "- **Teacher:** [meta-llama/Llama-3.2-3B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct) (3B parameters)\n", - "- **Student:** [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) (1B parameters)\n", - "\n", - "Both models share the same tokenizer/vocabulary (required for knowledge distillation) and include a chat template for deployment with `/chat/completions`.\n", - "\n", - "**HuggingFace Authentication:**\n", - "- For gated models (Llama, Gemma), you must provide a HuggingFace token via the `token_secret` parameter\n", - "- Get your token from [HuggingFace Settings](https://huggingface.co/settings/tokens) (requires Read access)\n", - "- Accept the model's terms on the HuggingFace model page before using it:\n", - " - [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct)\n", - " - [meta-llama/Llama-3.2-3B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct)" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "HF_TOKEN = os.getenv(\"HF_TOKEN\")\n", - "\n", - "\n", - "def create_or_get_secret(name: str, value: str | None, label: str):\n", - " if not value:\n", - " raise ValueError(f\"{label} is not set\")\n", - " try:\n", - " secret = client.secrets.create(\n", - " name=name,\n", - " workspace=\"default\",\n", - " value=value,\n", - " )\n", - " print(f\"Created secret: {name}\")\n", - " return secret\n", - " except ConflictError:\n", - " print(f\"Secret '{name}' already exists, continuing...\")\n", - " return client.secrets.retrieve(name=name, workspace=\"default\")\n", - "\n", - "\n", - "hf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\n", - "print(\"HF_TOKEN secret:\")\n", - "print(hf_secret.model_dump_json(indent=2))" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 4. Create Model FileSets and Model Entities\n", - "\n", - "Knowledge distillation requires **two** model entities:\n", - "1. **Student model** — the smaller model that will be trained ([meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct))\n", - "2. **Teacher model** — the larger model that provides soft targets ([meta-llama/Llama-3.2-3B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct))\n", - "\n", - "Using the Instruct variants ensures the output model includes a chat template, which is required for the `/chat/completions` inference endpoint." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from nemo_platform.types.files import HuggingfaceStorageConfigParam\n", - "\n", - "SPEC_TIMEOUT_SECONDS = 120\n", - "\n", - "\n", - "def create_model(hf_repo: str, model_name: str, description: str):\n", - " \"\"\"Create a fileset + model entity and wait for ModelSpec.\"\"\"\n", - " try:\n", - " client.files.filesets.create(\n", - " workspace=\"default\",\n", - " name=model_name,\n", - " description=description,\n", - " storage=HuggingfaceStorageConfigParam(\n", - " type=\"huggingface\",\n", - " repo_id=hf_repo,\n", - " repo_type=\"model\",\n", - " token_secret=hf_secret.name\n", - " )\n", - " )\n", - " print(f\"Created fileset: {model_name}\")\n", - " except ConflictError:\n", - " print(f\"Fileset '{model_name}' already exists.\")\n", - "\n", - " try:\n", - " model = client.models.create(\n", - " workspace=\"default\",\n", - " name=model_name,\n", - " fileset=f\"default/{model_name}\",\n", - " )\n", - " print(f\"Created Model Entity: {model_name}\")\n", - " except ConflictError:\n", - " print(f\"Model '{model_name}' already exists. Updating fileset.\")\n", - " model = client.models.update(\n", - " workspace=\"default\",\n", - " name=model_name,\n", - " fileset=f\"default/{model_name}\",\n", - " )\n", - "\n", - " print(f\"Waiting for ModelSpec on {model_name}...\")\n", - " spec_start = time.time()\n", - " while not model.spec:\n", - " if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n", - " raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS}s\")\n", - " time.sleep(2)\n", - " model = client.models.retrieve(workspace=\"default\", name=model_name)\n", - " print(f\"ModelSpec populated: {model.spec}\")\n", - " return model\n", - "\n", - "\n", - "student_model = create_model(\n", - " hf_repo=\"meta-llama/Llama-3.2-1B-Instruct\",\n", - " model_name=\"llama-3-2-1b-student\",\n", - " description=\"Llama 3.2 1B Instruct student model\",\n", - ")\n", - "\n", - "print()\n", - "\n", - "teacher_model = create_model(\n", - " hf_repo=\"meta-llama/Llama-3.2-3B-Instruct\",\n", - " model_name=\"llama-3-2-3b-teacher\",\n", - " description=\"Llama 3.2 3B Instruct teacher model\",\n", - ")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "---\n", - "\n", - "## Phase 1: Fine-Tune the Teacher\n", - "\n", - "### 5. Train Teacher with Full SFT\n", - "\n", - "For best distillation results, fine-tune the teacher on the **same dataset** that will be used for distillation. This ensures the teacher has learned the task-specific knowledge that the student will inherit.\n", - "\n", - "We train the 3B Instruct model with Full SFT on the SQuAD dataset." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from nemo_platform.types.customization import (\n", - " CustomizationJobInputParam,\n", - " SftTrainingParam,\n", - " DistillationTrainingParam,\n", - " ParallelismParamsParam,\n", - ")\n", - "\n", - "job_suffix = uuid.uuid4().hex[:4]\n", - "\n", - "TEACHER_JOB_NAME = f\"teacher-sft-job-{job_suffix}\"\n", - "\n", - "teacher_job = client.customization.jobs.create(\n", - " name=TEACHER_JOB_NAME,\n", - " workspace=\"default\",\n", - " spec=CustomizationJobInputParam(\n", - " model=f\"default/{teacher_model.name}\",\n", - " dataset=f\"fileset://default/{DATASET_NAME}\",\n", - " training=SftTrainingParam(\n", - " type=\"sft\",\n", - " epochs=1,\n", - " batch_size=64,\n", - " learning_rate=0.00005,\n", - " max_seq_length=2048,\n", - " micro_batch_size=1,\n", - " parallelism=ParallelismParamsParam(\n", - " num_gpus_per_node=1,\n", - " num_nodes=1,\n", - " tensor_parallel_size=1,\n", - " pipeline_parallel_size=1,\n", - " ),\n", - " ),\n", - " )\n", - ")\n", - "\n", - "TRAINED_TEACHER_NAME = teacher_job.spec.output.name\n", - "print(f\"Teacher training job: {teacher_job.name}\")\n", - "print(f\"Output teacher model: {TRAINED_TEACHER_NAME}\")" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from IPython.display import clear_output\n", - "\n", - "\n", - "def wait_for_job(job_name: str):\n", - " \"\"\"Poll job status until completion.\"\"\"\n", - " while True:\n", - " status = client.customization.jobs.get_status(name=job_name, workspace=\"default\")\n", - " clear_output(wait=True)\n", - " print(f\"Job: {job_name}\")\n", - " print(f\"Status: {status.status}\")\n", - "\n", - " for job_step in status.steps or []:\n", - " if job_step.name == \"customization-training-job\":\n", - " for task in job_step.tasks or []:\n", - " details = task.status_details or {}\n", - " step = details.get(\"step\")\n", - " max_steps = details.get(\"max_steps\")\n", - " if step is not None and max_steps is not None:\n", - " print(f\"Progress: Step {step}/{max_steps} ({step / max_steps * 100:.1f}%)\")\n", - " phase = details.get(\"phase\")\n", - " if phase:\n", - " print(f\"Phase: {phase}\")\n", - " break\n", - " break\n", - "\n", - " if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n", - " print(f\"\\nJob finished: {status.status}\")\n", - " return status\n", - "\n", - " time.sleep(10)\n", - "\n", - "\n", - "teacher_status = wait_for_job(TEACHER_JOB_NAME)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "---\n", - "\n", - "## Phase 2: Establish Baseline (Base Student)\n", - "\n", - "### 6. Deploy the Base Student Model\n", - "\n", - "Before distillation, deploy the base student model (1B Instruct, without any fine-tuning) to establish a baseline ROUGE score. After distillation, we compare the distilled student against this baseline to measure improvement." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from nemo_platform.types.inference import NIMDeploymentParam\n", - "\n", - "baseline_suffix = uuid.uuid4().hex[:4]\n", - "BASELINE_DEPLOYMENT_CONFIG = f\"baseline-student-cfg-{baseline_suffix}\"\n", - "BASELINE_DEPLOYMENT_NAME = f\"baseline-student-{baseline_suffix}\"\n", - "\n", - "baseline_deployment_config = client.inference.deployment_configs.create(\n", - " workspace=\"default\",\n", - " name=BASELINE_DEPLOYMENT_CONFIG,\n", - " nim_deployment=NIMDeploymentParam(\n", - " image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n", - " image_tag=\"1.15.5\",\n", - " gpu=1,\n", - " model_name=student_model.name,\n", - " model_namespace=\"default\",\n", - " additional_envs={\"NIM_MODEL_PROFILE\": \"vllm\"}\n", - " ),\n", - ")\n", - "\n", - "baseline_deployment = client.inference.deployments.create(\n", - " workspace=\"default\",\n", - " name=BASELINE_DEPLOYMENT_NAME,\n", - " config=baseline_deployment_config.name\n", - ")\n", - "\n", - "print(f\"Baseline student deployment: {baseline_deployment.name}\")" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "def wait_for_deployment(deployment_name: str, timeout_minutes: int = 30):\n", - " \"\"\"Poll deployment until ready.\"\"\"\n", - " start = time.time()\n", - " timeout = timeout_minutes * 60\n", - " while True:\n", - " dep = client.inference.deployments.retrieve(name=deployment_name, workspace=\"default\")\n", - " elapsed = time.time() - start\n", - " clear_output(wait=True)\n", - " print(f\"Deployment: {deployment_name}\")\n", - " print(f\"Status: {dep.status}\")\n", - " print(f\"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s\")\n", - "\n", - " if dep.status == \"READY\":\n", - " print(\"\\nDeployment is ready!\")\n", - " return dep\n", - " if dep.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n", - " print(f\"\\nDeployment failed: {dep.status}\")\n", - " return dep\n", - " if elapsed > timeout:\n", - " print(f\"\\nTimeout ({timeout_minutes}m). Check status manually.\")\n", - " return dep\n", - " time.sleep(15)\n", - "\n", - "\n", - "wait_for_deployment(BASELINE_DEPLOYMENT_NAME)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 7. Generate Baseline Predictions on Test Set" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "with open(f\"{DATASET_PATH}/testing.jsonl\", \"r\", encoding=\"utf-8\") as f:\n", - " test_data = [json.loads(line) for line in f]\n", - "\n", - "contexts = [row[\"context\"] for row in test_data]\n", - "questions = [row[\"question\"] for row in test_data]\n", - "reference_completions = [row[\"completion\"] for row in test_data]\n", - "\n", - "print(f\"Test samples: {len(contexts)}\")\n", - "print(f\"Sample context: {contexts[0]}\")\n", - "print(f\"Sample question: {questions[0]}\")\n", - "print(f\"Sample reference: {reference_completions[0]}\")" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "def generate_completions(\n", - " deployment_name: str,\n", - " output_model_name: str,\n", - " contexts: list[str],\n", - " questions: list[str],\n", - ") -> list[str]:\n", - " \"\"\"Generate completions for a list of context/question pairs using a deployed model.\"\"\"\n", - " completions = []\n", - " for context, question in zip(contexts, questions):\n", - " messages = [\n", - " {\n", - " \"role\": \"user\",\n", - " \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\",\n", - " }\n", - " ]\n", - " response = client.inference.gateway.provider.post(\n", - " \"v1/chat/completions\",\n", - " name=deployment_name,\n", - " workspace=\"default\",\n", - " body={\n", - " \"model\": f\"default/{output_model_name}\",\n", - " \"messages\": messages,\n", - " \"temperature\": 0,\n", - " \"max_tokens\": 128,\n", - " }\n", - " )\n", - " completions.append(response[\"choices\"][0][\"message\"][\"content\"])\n", - " return completions\n", - "\n", - "\n", - "print(\"Generating baseline (base student) predictions...\")\n", - "baseline_completions = generate_completions(BASELINE_DEPLOYMENT_NAME, student_model.name, contexts, questions)\n", - "print(f\"Generated {len(baseline_completions)} baseline predictions\")\n", - "print(f\"\\nSample baseline output: {baseline_completions[0]}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 8. Delete Baseline Deployment\n", - "\n", - "Delete the baseline student deployment to free GPU resources for the distillation training job and subsequent distilled model deployment." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace=\"default\")\n", - "print(f\"Deleted baseline deployment: {BASELINE_DEPLOYMENT_NAME}\")\n", - "\n", - "# wait for deployment to be deleted\n", - "time.sleep(60)\n", - "\n", - "client.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace=\"default\")\n", - "print(f\"Deleted baseline deployment config: {BASELINE_DEPLOYMENT_CONFIG}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "---\n", - "\n", - "## Phase 3: Distill into Student\n", - "\n", - "### 9. Create Knowledge Distillation Job\n", - "\n", - "Now create a distillation job that trains the 1B student using the **fine-tuned** 3B teacher's output distribution. The `model` field specifies the student, and `teacher_model` references the trained teacher model entity from Phase 1." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "**GPU Requirements:**\n", - "\n", - "KD requires loading both student and teacher models, so plan GPU memory accordingly:\n", - "- 1B student + 3B teacher: 1 GPU (24GB+ VRAM each)\n", - "- 3B student + 8B teacher: 4 GPUs\n", - "- 8B student + 70B teacher: 8+ GPUs\n", - "\n", - "Use `teacher_precision=\"bf16\"` (default) to reduce teacher memory footprint." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "KD_JOB_NAME = f\"my-kd-job-{job_suffix}\"\n", - "\n", - "kd_job = client.customization.jobs.create(\n", - " name=KD_JOB_NAME,\n", - " workspace=\"default\",\n", - " spec=CustomizationJobInputParam(\n", - " model=f\"default/{student_model.name}\",\n", - " dataset=f\"fileset://default/{DATASET_NAME}\",\n", - " training=DistillationTrainingParam(\n", - " type=\"distillation\",\n", - " teacher_model=f\"default/{TRAINED_TEACHER_NAME}\",\n", - " teacher_precision=\"bf16\",\n", - " distillation_ratio=0.5,\n", - " distillation_temperature=2.0,\n", - " epochs=1,\n", - " batch_size=64,\n", - " learning_rate=0.00005,\n", - " max_seq_length=2048,\n", - " micro_batch_size=1,\n", - " parallelism=ParallelismParamsParam(\n", - " num_gpus_per_node=1,\n", - " num_nodes=1,\n", - " tensor_parallel_size=1,\n", - " pipeline_parallel_size=1,\n", - " ),\n", - " ),\n", - " )\n", - ")\n", - "\n", - "DISTILLED_STUDENT_NAME = kd_job.spec.output.name\n", - "print(f\"Distillation job: {kd_job.name}\")\n", - "print(f\"Output student model: {DISTILLED_STUDENT_NAME}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 10. Track Distillation Progress" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "kd_status = wait_for_job(KD_JOB_NAME)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "---\n", - "\n", - "## Phase 4: Evaluate the Distilled Student Model\n", - "\n", - "### 11. Deploy the Distilled Student Model\n", - "\n", - "The output model has the same architecture as the 1B student—only its weights have been updated via distillation. It requires just 1 GPU to deploy." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "deploy_suffix_2 = uuid.uuid4().hex[:4]\n", - "STUDENT_DEPLOYMENT_CONFIG = f\"kd-student-deploy-cfg-{deploy_suffix_2}\"\n", - "STUDENT_DEPLOYMENT_NAME = f\"kd-student-deploy-{deploy_suffix_2}\"\n", - "\n", - "student_deployment_config = client.inference.deployment_configs.create(\n", - " workspace=\"default\",\n", - " name=STUDENT_DEPLOYMENT_CONFIG,\n", - " nim_deployment=NIMDeploymentParam(\n", - " image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n", - " image_tag=\"1.15.5\",\n", - " gpu=1,\n", - " model_name=DISTILLED_STUDENT_NAME,\n", - " model_namespace=\"default\",\n", - " additional_envs={\"NIM_MODEL_PROFILE\": \"vllm\"},\n", - " ),\n", - ")\n", - "\n", - "student_deployment = client.inference.deployments.create(\n", - " workspace=\"default\",\n", - " name=STUDENT_DEPLOYMENT_NAME,\n", - " config=student_deployment_config.name\n", - ")\n", - "\n", - "print(f\"Student deployment: {student_deployment.name}\")" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "wait_for_deployment(STUDENT_DEPLOYMENT_NAME)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 12. Generate Student Predictions on Test Set" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "print(\"Generating distilled student predictions...\")\n", - "student_completions = generate_completions(STUDENT_DEPLOYMENT_NAME, DISTILLED_STUDENT_NAME, contexts, questions)\n", - "print(f\"Generated {len(student_completions)} student predictions\")\n", - "print(f\"\\nSample student output: {student_completions[0]}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 13. Compute ROUGE Scores\n", - "\n", - "Compare the base student (before distillation) and the distilled student against the ground-truth reference completions using ROUGE metrics." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import evaluate\n", - "\n", - "rouge = evaluate.load(\"rouge\")\n", - "\n", - "baseline_scores = rouge.compute(predictions=baseline_completions, references=reference_completions)\n", - "student_scores = rouge.compute(predictions=student_completions, references=reference_completions)\n", - "\n", - "metrics = list(baseline_scores.keys())\n", - "header = f\"{'Model':<35} \" + \" \".join(f\"{m:>10}\" for m in metrics)\n", - "separator = \"-\" * len(header)\n", - "\n", - "print(\"=\" * 60)\n", - "print(\"ROUGE SCORE COMPARISON\")\n", - "print(\"=\" * 60)\n", - "print(header)\n", - "print(separator)\n", - "print(f\"{'Base Student (1B, no training)':<35} \" + \" \".join(f\"{baseline_scores[m]:>10.4f}\" for m in metrics))\n", - "print(f\"{'Distilled Student (1B, KD)':<35} \" + \" \".join(f\"{student_scores[m]:>10.4f}\" for m in metrics))" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "print(\"Sample predictions (first 3):\\n\")\n", - "for i in range(min(3, len(contexts))):\n", - " print(f\"--- Sample {i + 1} ---\")\n", - " print(f\"Question: {questions[i]}\")\n", - " print(f\"Reference: {reference_completions[i]}\")\n", - " print(f\"Baseline: {baseline_completions[i][:200]}\")\n", - " print(f\"Distilled: {student_completions[i][:200]}\")\n", - " print()" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "**Interpreting ROUGE Scores:**\n", - "\n", - "| Metric | Measures |\n", - "|--------|----------|\n", - "| **ROUGE-1** | Unigram overlap between prediction and reference |\n", - "| **ROUGE-2** | Bigram overlap (captures phrase-level similarity) |\n", - "| **ROUGE-L** | Longest common subsequence (captures sentence structure) |\n", - "| **ROUGE-Lsum** | ROUGE-L computed over full summaries |\n", - "\n", - "**What to expect:**\n", - "- The base student (1B, no training) provides a lower bound since it has not seen the task data\n", - "- The distilled student (1B, KD) should significantly outperform the base student, demonstrating the knowledge transferred from the 3B teacher\n", - "- If the distilled student scores are not much higher than the baseline, try increasing `distillation_temperature`, adjusting `distillation_ratio`, or training for more epochs\n", - "\n", - "---\n", - "\n", - "## Hyperparameters\n", - "\n", - "For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n", - "\n", - "---\n", - "\n", - "## Troubleshooting\n", - "\n", - "**Job fails during model download:**\n", - "- Verify authentication secrets are configured (refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n", - "- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n", - "- Check both `model` (student) and `teacher_model` URNs are correct\n", - "- Ensure both model entities exist: `client.models.retrieve(name=..., workspace=\"default\")`\n", - "\n", - "**Job fails with OOM (Out of Memory) error:**\n", - "\n", - "KD loads both models, so OOM is more likely than with SFT:\n", - "1. **First try:** Use `teacher_precision=\"bf16\"` to reduce teacher memory\n", - "2. **Still OOM:** Reduce `micro_batch_size` to 1\n", - "3. **Still OOM:** Reduce `batch_size` and `max_seq_length`\n", - "4. **Last resort:** Increase `num_gpus_per_node`\n", - "\n", - "**No chat template / `/chat/completions` fails:**\n", - "- Use Instruct model variants (e.g., `Llama-3.2-1B-Instruct`) instead of base models (`Llama-3.2-1B`). Base models do not include a chat template in their tokenizer, so the output model will also lack one.\n", - "\n", - "**Distilled model quality is poor:**\n", - "- Increase `distillation_temperature` (try 2.0–5.0) to transfer more nuanced knowledge\n", - "- Adjust `distillation_ratio`—if dataset labels are high-quality, lower the ratio; if the teacher is strong, raise it\n", - "- Increase `epochs` or `max_steps` for more training\n", - "- Verify teacher and student share the same vocabulary\n", - "\n", - "**Vocabulary mismatch error:**\n", - "- Teacher and student must use the same tokenizer. Use models from the same family (e.g., Llama 3.2 1B Instruct + Llama 3.2 3B Instruct)\n", - "\n", - "**Deployment fails:**\n", - "- Verify output model exists: `client.models.retrieve(name=job.spec.output.name, workspace=\"default\")`\n", - "- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n", - "- The distilled model has the same size as the student, so GPU requirements match the student model\n", - "\n", - "\n", - "## Next Steps\n", - "\n", - "- [Monitor training metrics](fine-tune-metrics) in detail\n", - "- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n", - "- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning\n", - "- Learn about [Full SFT](./sft-customization-job) for direct supervised fine-tuning" - ] - } - ], - "metadata": { - "kernelspec": { - "display_name": ".venv", - "language": "python", - "name": "python3" - }, - "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 3 - }, - "file_extension": ".py", - "mimetype": "text/x-python", - "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython3", - "version": "3.11.14" - } - }, - "nbformat": 4, - "nbformat_minor": 4 + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "\n", + "\n", + "# Knowledge Distillation Customization\n", + "\n", + "Learn how to train a smaller student model to mimic a larger teacher model using knowledge distillation (KD).\n", + "\n", + "## About\n", + "\n", + "Knowledge Distillation transfers knowledge from a large **teacher** model to a smaller **student** model. During training, the student learns to match the teacher's output probability distribution, producing a compact model that retains much of the teacher's capability.\n", + "\n", + "**What you can achieve with KD:**\n", + "\n", + "- **Compress models:** Distill a 3B model into a 1B model for faster inference and lower deployment costs\n", + "- **Reduce latency:** Deploy a smaller model that responds faster while preserving quality\n", + "- **Lower resource requirements:** Serve a distilled model on fewer GPUs\n", + "\n", + "### KD vs SFT: Understanding the Trade-offs\n", + "\n", + "| Aspect | Full SFT | Knowledge Distillation |\n", + "| --- | --- | --- |\n", + "| **Training signal** | Ground-truth labels only | Teacher's soft probability distribution + labels |\n", + "| **Knowledge source** | Dataset examples | Teacher model's learned representations |\n", + "| **Output model size** | Same as input model | Typically a smaller student model |\n", + "| **GPU requirements** | Needs to fit one model | Needs to fit both teacher and student in memory |\n", + "| **Best for** | Domain adaptation, new knowledge injection | Model compression, latency reduction |\n", + "\n", + "### Key Parameters\n", + "\n", + "| Parameter | Default | Description |\n", + "| --- | --- | --- |\n", + "| `teacher_model` | *(required)* | Teacher model entity URN (e.g., `default/llama-3-2-3b-teacher`) |\n", + "| `teacher_precision` | `bf16` | Precision for the frozen teacher (`bf16`, `fp16`, `fp32`). Lower = less memory |\n", + "| `distillation_ratio` | `0.5` | Balance between CE loss and KD loss. `0.0` = CE only, `1.0` = KD only |\n", + "| `distillation_temperature` | `1.0` | Softmax temperature. Higher = softer distributions, more knowledge transfer |\n", + "\n", + "### Workflow Overview\n", + "\n", + "This tutorial follows a complete distillation pipeline:\n", + "\n", + "1. **Fine-tune the teacher** (SFT on the task dataset) so it learns the domain\n", + "2. **Establish a baseline** by deploying the base student model and measuring ROUGE scores\n", + "3. **Distill into the student** using the fine-tuned teacher's soft targets\n", + "4. **Evaluate the distilled student** and compare ROUGE scores against the baseline\n", + "\n", + "**When to choose KD:**\n", + "\n", + "- You have a high-quality large model and want a smaller, faster version\n", + "- Deployment latency or cost is a constraint\n", + "- The teacher and student share the same vocabulary (e.g., both are Llama models)\n", + "\n", + "**When to choose SFT instead:** Refer to the [Full SFT tutorial](./sft-customization-job) when you want to train a model directly on labeled data without a teacher." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Prerequisites\n", + "\n", + "Before starting this tutorial, ensure you have:\n", + "\n", + "1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n", + "2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n", + "3. **Installed evaluation dependencies:**\n", + "\n", + "```sh\n", + "pip install evaluate rouge_score datasets\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Quick Start\n", + "\n", + "### 1. Initialize SDK\n", + "\n", + "The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](../../get-started/quickstart.md) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable:\n", + "\n", + "```sh\n", + "export NMP_BASE_URL=\n", + "```" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import time\n", + "import uuid\n", + "from pathlib import Path\n", + "\n", + "from nemo_platform import NeMoPlatform, ConflictError\n", + "\n", + "NMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\n", + "client = NeMoPlatform(\n", + " base_url=NMP_BASE_URL,\n", + " workspace=\"default\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2. Prepare Dataset\n", + "\n", + "Knowledge distillation uses the same dataset formats as SFT. We use the SQuAD dataset for both teacher training and distillation so that the teacher first learns the task, then transfers that knowledge to the student.\n", + "\n", + "We also hold out a small **test split** for ROUGE evaluation at the end." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from datasets import load_dataset, DatasetDict\n", + "\n", + "print(\"Loading dataset rajpurkar/squad\")\n", + "raw_dataset = load_dataset(\"rajpurkar/squad\")\n", + "if not isinstance(raw_dataset, DatasetDict):\n", + " raise ValueError(\"Dataset does not contain expected splits\")\n", + "\n", + "print(\"Loaded dataset\")\n", + "\n", + "SEED = 1234\n", + "TRAINING_SIZE = 3000\n", + "VALIDATION_SIZE = 300\n", + "TEST_SIZE = 100\n", + "DATASET_PATH = Path(\"kd-dataset\").absolute()\n", + "\n", + "os.makedirs(DATASET_PATH, exist_ok=True)\n", + "\n", + "train_set = raw_dataset.get('train')\n", + "split = train_set.train_test_split(test_size=0.05, seed=SEED)\n", + "\n", + "train_ds = split['train'].select(range(min(TRAINING_SIZE, len(split['train']))))\n", + "val_ds = split['test'].select(range(min(VALIDATION_SIZE, len(split['test']))))\n", + "test_ds = split['test'].select(range(VALIDATION_SIZE, min(VALIDATION_SIZE + TEST_SIZE, len(split['test']))))\n", + "\n", + "\n", + "def convert_squad(example):\n", + " \"\"\"Convert SQuAD format to prompt/completion format.\"\"\"\n", + " prompt = f\"Context: {example['context']} Question: {example['question']} Answer:\"\n", + " completion = example[\"answers\"][\"text\"][0]\n", + " return {\"prompt\": prompt, \"completion\": completion}\n", + "\n", + "\n", + "def write_jsonl(dataset, path):\n", + " with open(path, \"w\", encoding=\"utf-8\") as f:\n", + " for example in dataset:\n", + " f.write(json.dumps(convert_squad(example)) + \"\\n\")\n", + "\n", + "\n", + "def write_test_jsonl(dataset, path):\n", + " \"\"\"Save test split with raw context/question for chat-style evaluation.\"\"\"\n", + " with open(path, \"w\", encoding=\"utf-8\") as f:\n", + " for example in dataset:\n", + " f.write(json.dumps({\n", + " \"context\": example[\"context\"],\n", + " \"question\": example[\"question\"],\n", + " \"completion\": example[\"answers\"][\"text\"][0],\n", + " }) + \"\\n\")\n", + "\n", + "\n", + "write_jsonl(train_ds, f\"{DATASET_PATH}/training.jsonl\")\n", + "write_jsonl(val_ds, f\"{DATASET_PATH}/validation.jsonl\")\n", + "write_test_jsonl(test_ds, f\"{DATASET_PATH}/testing.jsonl\")\n", + "\n", + "print(f\"Training: {len(train_ds)} rows\")\n", + "print(f\"Validation: {len(val_ds)} rows\")\n", + "print(f\"Test: {len(test_ds)} rows\")\n", + "\n", + "with open(f\"{DATASET_PATH}/training.jsonl\", 'r') as f:\n", + " sample = json.loads(f.readline())\n", + " print(f\"\\nSample prompt: {sample['prompt'][:150]}...\")\n", + " print(f\"Sample completion: {sample['completion']}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "DATASET_NAME = \"kd-dataset\"\n", + "\n", + "try:\n", + " client.files.filesets.create(\n", + " workspace=\"default\",\n", + " name=DATASET_NAME,\n", + " description=\"Knowledge distillation training data\"\n", + " )\n", + " print(f\"Created fileset: {DATASET_NAME}\")\n", + "except ConflictError:\n", + " print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n", + "\n", + "client.files.upload(\n", + " local_path=f\"{DATASET_PATH}/\",\n", + " remote_path=\"\",\n", + " fileset=DATASET_NAME,\n", + " workspace=\"default\"\n", + ")\n", + "\n", + "print(\"Uploaded files:\")\n", + "print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 3. Secrets Setup\n", + "\n", + "In this tutorial we use two Llama 3.2 Instruct models from HuggingFace:\n", + "- **Teacher:** [meta-llama/Llama-3.2-3B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct) (3B parameters)\n", + "- **Student:** [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) (1B parameters)\n", + "\n", + "Both models share the same tokenizer/vocabulary (required for knowledge distillation) and include a chat template for deployment with `/chat/completions`.\n", + "\n", + "**HuggingFace Authentication:**\n", + "- For gated models (Llama, Gemma), you must provide a HuggingFace token via the `token_secret` parameter\n", + "- Get your token from [HuggingFace Settings](https://huggingface.co/settings/tokens) (requires Read access)\n", + "- Accept the model's terms on the HuggingFace model page before using it:\n", + " - [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct)\n", + " - [meta-llama/Llama-3.2-3B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "HF_TOKEN = os.getenv(\"HF_TOKEN\")\n", + "\n", + "\n", + "def create_or_get_secret(name: str, value: str | None, label: str):\n", + " if not value:\n", + " raise ValueError(f\"{label} is not set\")\n", + " try:\n", + " secret = client.secrets.create(\n", + " name=name,\n", + " workspace=\"default\",\n", + " value=value,\n", + " )\n", + " print(f\"Created secret: {name}\")\n", + " return secret\n", + " except ConflictError:\n", + " print(f\"Secret '{name}' already exists, continuing...\")\n", + " return client.secrets.retrieve(name=name, workspace=\"default\")\n", + "\n", + "\n", + "hf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\n", + "print(\"HF_TOKEN secret:\")\n", + "print(hf_secret.model_dump_json(indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 4. Create Model FileSets and Model Entities\n", + "\n", + "Knowledge distillation requires **two** model entities:\n", + "1. **Student model** — the smaller model that will be trained ([meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct))\n", + "2. **Teacher model** — the larger model that provides soft targets ([meta-llama/Llama-3.2-3B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct))\n", + "\n", + "Using the Instruct variants ensures the output model includes a chat template, which is required for the `/chat/completions` inference endpoint." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from nemo_platform.types.files import HuggingfaceStorageConfigParam\n", + "\n", + "SPEC_TIMEOUT_SECONDS = 120\n", + "\n", + "\n", + "def create_model(hf_repo: str, model_name: str, description: str):\n", + " \"\"\"Create a fileset + model entity and wait for ModelSpec.\"\"\"\n", + " try:\n", + " client.files.filesets.create(\n", + " workspace=\"default\",\n", + " name=model_name,\n", + " description=description,\n", + " storage=HuggingfaceStorageConfigParam(\n", + " type=\"huggingface\",\n", + " repo_id=hf_repo,\n", + " repo_type=\"model\",\n", + " token_secret=hf_secret.name\n", + " )\n", + " )\n", + " print(f\"Created fileset: {model_name}\")\n", + " except ConflictError:\n", + " print(f\"Fileset '{model_name}' already exists.\")\n", + "\n", + " try:\n", + " model = client.models.create(\n", + " workspace=\"default\",\n", + " name=model_name,\n", + " fileset=f\"default/{model_name}\",\n", + " )\n", + " print(f\"Created Model Entity: {model_name}\")\n", + " except ConflictError:\n", + " print(f\"Model '{model_name}' already exists. Updating fileset.\")\n", + " model = client.models.update(\n", + " workspace=\"default\",\n", + " name=model_name,\n", + " fileset=f\"default/{model_name}\",\n", + " )\n", + "\n", + " print(f\"Waiting for ModelSpec on {model_name}...\")\n", + " spec_start = time.time()\n", + " while not model.spec:\n", + " if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n", + " raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS}s\")\n", + " time.sleep(2)\n", + " model = client.models.retrieve(workspace=\"default\", name=model_name)\n", + " print(f\"ModelSpec populated: {model.spec}\")\n", + " return model\n", + "\n", + "\n", + "student_model = create_model(\n", + " hf_repo=\"meta-llama/Llama-3.2-1B-Instruct\",\n", + " model_name=\"llama-3-2-1b-student\",\n", + " description=\"Llama 3.2 1B Instruct student model\",\n", + ")\n", + "\n", + "print()\n", + "\n", + "teacher_model = create_model(\n", + " hf_repo=\"meta-llama/Llama-3.2-3B-Instruct\",\n", + " model_name=\"llama-3-2-3b-teacher\",\n", + " description=\"Llama 3.2 3B Instruct teacher model\",\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "---\n", + "\n", + "## Phase 1: Fine-Tune the Teacher\n", + "\n", + "### 5. Train Teacher with Full SFT\n", + "\n", + "For best distillation results, fine-tune the teacher on the **same dataset** that will be used for distillation. This ensures the teacher has learned the task-specific knowledge that the student will inherit.\n", + "\n", + "We train the 3B Instruct model with Full SFT on the SQuAD dataset." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from nemo_automodel_plugin.schema import AutomodelJobInput\n", + "\n", + "job_suffix = uuid.uuid4().hex[:4]\n", + "\n", + "TEACHER_JOB_NAME = f\"teacher-sft-job-{job_suffix}\"\n", + "TEACHER_OUTPUT_NAME = f\"teacher-model-{job_suffix}\"\n", + "\n", + "teacher_spec = AutomodelJobInput(\n", + " model=f\"default/{teacher_model.name}\",\n", + " dataset={\"training\": f\"default/{DATASET_NAME}\"},\n", + " training={\n", + " \"training_type\": \"sft\",\n", + " \"finetuning_type\": \"all_weights\",\n", + " \"max_seq_length\": 2048,\n", + " },\n", + " schedule={\"epochs\": 1},\n", + " batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n", + " optimizer={\"learning_rate\": 5e-5},\n", + " parallelism={\"num_gpus_per_node\": 1},\n", + " output={\"name\": TEACHER_OUTPUT_NAME},\n", + ")\n", + "\n", + "teacher_job = client.customization.automodel.jobs.create(\n", + " spec=teacher_spec, workspace=\"default\", name=TEACHER_JOB_NAME\n", + ")\n", + "\n", + "TRAINED_TEACHER_NAME = TEACHER_OUTPUT_NAME\n", + "print(f\"Teacher training job: {teacher_job.job.name}\")\n", + "print(f\"Output teacher model: {TRAINED_TEACHER_NAME}\")\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from IPython.display import clear_output\n", + "\n", + "\n", + "def wait_for_job(job_name: str):\n", + " \"\"\"Poll job status until completion.\"\"\"\n", + " while True:\n", + " status = client.jobs.get_status(name=job_name, workspace=\"default\")\n", + " clear_output(wait=True)\n", + " print(f\"Job: {job_name}\")\n", + " print(f\"Status: {status.status}\")\n", + "\n", + " for job_step in status.steps or []:\n", + " if job_step.name == \"training\":\n", + " for task in job_step.tasks or []:\n", + " details = task.status_details or {}\n", + " step = details.get(\"step\")\n", + " max_steps = details.get(\"max_steps\")\n", + " if step is not None and max_steps is not None:\n", + " print(f\"Progress: Step {step}/{max_steps} ({step / max_steps * 100:.1f}%)\")\n", + " phase = details.get(\"phase\")\n", + " if phase:\n", + " print(f\"Phase: {phase}\")\n", + " break\n", + " break\n", + "\n", + " if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n", + " print(f\"\\nJob finished: {status.status}\")\n", + " return status\n", + "\n", + " time.sleep(10)\n", + "\n", + "\n", + "teacher_status = wait_for_job(TEACHER_JOB_NAME)\n", + "assert teacher_status.status == \"completed\"" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "---\n", + "\n", + "## Phase 2: Establish Baseline (Base Student)\n", + "\n", + "### 6. Deploy the Base Student Model\n", + "\n", + "Before distillation, deploy the base student model (1B Instruct, without any fine-tuning) to establish a baseline ROUGE score. After distillation, we compare the distilled student against this baseline to measure improvement." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "baseline_suffix = uuid.uuid4().hex[:4]\n", + "BASELINE_DEPLOYMENT_CONFIG = f\"baseline-student-cfg-{baseline_suffix}\"\n", + "BASELINE_DEPLOYMENT_NAME = f\"baseline-student-{baseline_suffix}\"\n", + "\n", + "baseline_deployment_config = client.inference.deployment_configs.create(\n", + " workspace=\"default\",\n", + " name=BASELINE_DEPLOYMENT_CONFIG,\n", + " engine=\"vllm\",\n", + " model_spec={\n", + " \"model_namespace\": \"default\",\n", + " \"model_name\": student_model.name,\n", + " },\n", + " executor_config={\n", + " \"gpu\": 1,\n", + " \"image_name\": \"vllm/vllm-openai\",\n", + " \"image_tag\": \"v0.22.1\",\n", + " },\n", + ")\n", + "\n", + "baseline_deployment = client.inference.deployments.create(\n", + " workspace=\"default\",\n", + " name=BASELINE_DEPLOYMENT_NAME,\n", + " config=baseline_deployment_config.name\n", + ")\n", + "\n", + "print(f\"Baseline student deployment: {baseline_deployment.name}\")\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def wait_for_deployment(deployment_name: str, timeout_minutes: int = 30):\n", + " \"\"\"Poll deployment until ready.\"\"\"\n", + " start = time.time()\n", + " timeout = timeout_minutes * 60\n", + " while True:\n", + " dep = client.inference.deployments.retrieve(name=deployment_name, workspace=\"default\")\n", + " elapsed = time.time() - start\n", + " clear_output(wait=True)\n", + " print(f\"Deployment: {deployment_name}\")\n", + " print(f\"Status: {dep.status}\")\n", + " print(f\"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s\")\n", + "\n", + " if dep.status == \"READY\":\n", + " print(\"\\nDeployment is ready!\")\n", + " return dep\n", + " if dep.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n", + " print(f\"\\nDeployment failed: {dep.status}\")\n", + " return dep\n", + " if elapsed > timeout:\n", + " print(f\"\\nTimeout ({timeout_minutes}m). Check status manually.\")\n", + " return dep\n", + " time.sleep(15)\n", + "\n", + "\n", + "dep_status = wait_for_deployment(BASELINE_DEPLOYMENT_NAME)\n", + "assert dep_status.status == \"READY\"" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 7. Generate Baseline Predictions on Test Set" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "with open(f\"{DATASET_PATH}/testing.jsonl\", \"r\", encoding=\"utf-8\") as f:\n", + " test_data = [json.loads(line) for line in f]\n", + "\n", + "contexts = [row[\"context\"] for row in test_data]\n", + "questions = [row[\"question\"] for row in test_data]\n", + "reference_completions = [row[\"completion\"] for row in test_data]\n", + "\n", + "print(f\"Test samples: {len(contexts)}\")\n", + "print(f\"Sample context: {contexts[0]}\")\n", + "print(f\"Sample question: {questions[0]}\")\n", + "print(f\"Sample reference: {reference_completions[0]}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "def generate_completions(\n", + " deployment_name: str,\n", + " output_model_name: str,\n", + " contexts: list[str],\n", + " questions: list[str],\n", + ") -> list[str]:\n", + " \"\"\"Generate completions for a list of context/question pairs using a deployed model.\"\"\"\n", + " completions = []\n", + " for context, question in zip(contexts, questions):\n", + " messages = [\n", + " {\n", + " \"role\": \"user\",\n", + " \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\",\n", + " }\n", + " ]\n", + " response = client.inference.gateway.provider.post(\n", + " \"v1/chat/completions\",\n", + " name=deployment_name,\n", + " workspace=\"default\",\n", + " body={\n", + " \"model\": f\"default/{output_model_name}\",\n", + " \"messages\": messages,\n", + " \"temperature\": 0,\n", + " \"max_tokens\": 128,\n", + " }\n", + " )\n", + " completions.append(response[\"choices\"][0][\"message\"][\"content\"])\n", + " return completions\n", + "\n", + "\n", + "print(\"Generating baseline (base student) predictions...\")\n", + "baseline_completions = generate_completions(BASELINE_DEPLOYMENT_NAME, student_model.name, contexts, questions)\n", + "print(f\"Generated {len(baseline_completions)} baseline predictions\")\n", + "print(f\"\\nSample baseline output: {baseline_completions[0]}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 8. Delete Baseline Deployment\n", + "\n", + "Delete the baseline student deployment to free GPU resources for the distillation training job and subsequent distilled model deployment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace=\"default\")\n", + "print(f\"Deleted baseline deployment: {BASELINE_DEPLOYMENT_NAME}\")\n", + "\n", + "if not client.models.wait_for_status(\n", + " deployment_name=BASELINE_DEPLOYMENT_NAME,\n", + " desired_status=\"DELETED\",\n", + " workspace=\"default\",\n", + " timeout=600,\n", + "):\n", + " raise TimeoutError(\n", + " f\"Deployment {BASELINE_DEPLOYMENT_NAME} was not deleted within timeout\"\n", + " )\n", + "\n", + "client.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace=\"default\")\n", + "print(f\"Deleted baseline deployment config: {BASELINE_DEPLOYMENT_CONFIG}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "---\n", + "\n", + "## Phase 3: Distill into Student\n", + "\n", + "### 9. Create Knowledge Distillation Job\n", + "\n", + "Now create a distillation job that trains the 1B student using the **fine-tuned** 3B teacher's output distribution. The `model` field specifies the student, and `teacher_model` references the trained teacher model entity from Phase 1." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "**GPU Requirements:**\n", + "\n", + "KD requires loading both student and teacher models, so plan GPU memory accordingly:\n", + "- 1B student + 3B teacher: 1 GPU (24GB+ VRAM each)\n", + "- 3B student + 8B teacher: 4 GPUs\n", + "- 8B student + 70B teacher: 8+ GPUs\n", + "\n", + "Use `teacher_precision=\"bf16\"` (default) to reduce teacher memory footprint." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from nemo_automodel_plugin.schema import AutomodelJobInput\n", + "\n", + "KD_JOB_NAME = f\"my-kd-job-{job_suffix}\"\n", + "KD_OUTPUT_NAME = f\"kd-student-{job_suffix}\"\n", + "\n", + "kd_spec = AutomodelJobInput(\n", + " model=f\"default/{student_model.name}\",\n", + " dataset={\"training\": f\"default/{DATASET_NAME}\"},\n", + " training={\n", + " \"training_type\": \"distillation\",\n", + " \"finetuning_type\": \"all_weights\",\n", + " \"teacher_model\": f\"default/{TRAINED_TEACHER_NAME}\",\n", + " \"teacher_precision\": \"bf16\",\n", + " \"distillation_ratio\": 0.5,\n", + " \"distillation_temperature\": 2.0,\n", + " \"max_seq_length\": 2048,\n", + " },\n", + " schedule={\"epochs\": 1},\n", + " batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n", + " optimizer={\"learning_rate\": 5e-5},\n", + " parallelism={\"num_gpus_per_node\": 1},\n", + " output={\"name\": KD_OUTPUT_NAME},\n", + ")\n", + "\n", + "kd_job = client.customization.automodel.jobs.create(\n", + " spec=kd_spec, workspace=\"default\", name=KD_JOB_NAME\n", + ")\n", + "\n", + "DISTILLED_STUDENT_NAME = KD_OUTPUT_NAME\n", + "print(f\"Distillation job: {kd_job.job.name}\")\n", + "print(f\"Output student model: {DISTILLED_STUDENT_NAME}\")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 10. Track Distillation Progress" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "kd_status = wait_for_job(KD_JOB_NAME)\n", + "assert kd_status.status == \"completed\"" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "---\n", + "\n", + "## Phase 4: Evaluate the Distilled Student Model\n", + "\n", + "### 11. Deploy the Distilled Student Model\n", + "\n", + "The output model has the same architecture as the 1B student—only its weights have been updated via distillation. It requires just 1 GPU to deploy." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "deploy_suffix_2 = uuid.uuid4().hex[:4]\n", + "STUDENT_DEPLOYMENT_CONFIG = f\"kd-student-deploy-cfg-{deploy_suffix_2}\"\n", + "STUDENT_DEPLOYMENT_NAME = f\"kd-student-deploy-{deploy_suffix_2}\"\n", + "\n", + "student_deployment_config = client.inference.deployment_configs.create(\n", + " workspace=\"default\",\n", + " name=STUDENT_DEPLOYMENT_CONFIG,\n", + " engine=\"vllm\",\n", + " model_spec={\n", + " \"model_namespace\": \"default\",\n", + " \"model_name\": DISTILLED_STUDENT_NAME,\n", + " },\n", + " executor_config={\n", + " \"gpu\": 1,\n", + " \"image_name\": \"vllm/vllm-openai\",\n", + " \"image_tag\": \"v0.22.1\",\n", + " },\n", + ")\n", + "\n", + "student_deployment = client.inference.deployments.create(\n", + " workspace=\"default\",\n", + " name=STUDENT_DEPLOYMENT_NAME,\n", + " config=student_deployment_config.name\n", + ")\n", + "\n", + "print(f\"Student deployment: {student_deployment.name}\")\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "wait_for_deployment(STUDENT_DEPLOYMENT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 12. Generate Student Predictions on Test Set" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "print(\"Generating distilled student predictions...\")\n", + "student_completions = generate_completions(STUDENT_DEPLOYMENT_NAME, DISTILLED_STUDENT_NAME, contexts, questions)\n", + "print(f\"Generated {len(student_completions)} student predictions\")\n", + "print(f\"\\nSample student output: {student_completions[0]}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 13. Compute ROUGE Scores\n", + "\n", + "Compare the base student (before distillation) and the distilled student against the ground-truth reference completions using ROUGE metrics." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import evaluate\n", + "\n", + "rouge = evaluate.load(\"rouge\")\n", + "\n", + "baseline_scores = rouge.compute(predictions=baseline_completions, references=reference_completions)\n", + "student_scores = rouge.compute(predictions=student_completions, references=reference_completions)\n", + "\n", + "metrics = list(baseline_scores.keys())\n", + "header = f\"{'Model':<35} \" + \" \".join(f\"{m:>10}\" for m in metrics)\n", + "separator = \"-\" * len(header)\n", + "\n", + "print(\"=\" * 60)\n", + "print(\"ROUGE SCORE COMPARISON\")\n", + "print(\"=\" * 60)\n", + "print(header)\n", + "print(separator)\n", + "print(f\"{'Base Student (1B, no training)':<35} \" + \" \".join(f\"{baseline_scores[m]:>10.4f}\" for m in metrics))\n", + "print(f\"{'Distilled Student (1B, KD)':<35} \" + \" \".join(f\"{student_scores[m]:>10.4f}\" for m in metrics))" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "print(\"Sample predictions (first 3):\\n\")\n", + "for i in range(min(3, len(contexts))):\n", + " print(f\"--- Sample {i + 1} ---\")\n", + " print(f\"Question: {questions[i]}\")\n", + " print(f\"Reference: {reference_completions[i]}\")\n", + " print(f\"Baseline: {baseline_completions[i][:200]}\")\n", + " print(f\"Distilled: {student_completions[i][:200]}\")\n", + " print()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "**Interpreting ROUGE Scores:**\n", + "\n", + "| Metric | Measures |\n", + "|--------|----------|\n", + "| **ROUGE-1** | Unigram overlap between prediction and reference |\n", + "| **ROUGE-2** | Bigram overlap (captures phrase-level similarity) |\n", + "| **ROUGE-L** | Longest common subsequence (captures sentence structure) |\n", + "| **ROUGE-Lsum** | ROUGE-L computed over full summaries |\n", + "\n", + "**What to expect:**\n", + "- The base student (1B, no training) provides a lower bound since it has not seen the task data\n", + "- The distilled student (1B, KD) should significantly outperform the base student, demonstrating the knowledge transferred from the 3B teacher\n", + "- If the distilled student scores are not much higher than the baseline, try increasing `distillation_temperature`, adjusting `distillation_ratio`, or training for more epochs\n", + "\n", + "---\n", + "\n", + "## Hyperparameters\n", + "\n", + "For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n", + "\n", + "---\n", + "\n", + "## Troubleshooting\n", + "\n", + "**Job fails during model download:**\n", + "- Verify authentication secrets are configured (refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n", + "- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n", + "- Check both `model` (student) and `teacher_model` URNs are correct\n", + "- Ensure both model entities exist: `client.models.retrieve(name=..., workspace=\"default\")`\n", + "\n", + "**Job fails with OOM (Out of Memory) error:**\n", + "\n", + "KD loads both models, so OOM is more likely than with SFT:\n", + "1. **First try:** Use `teacher_precision=\"bf16\"` to reduce teacher memory\n", + "2. **Still OOM:** Reduce `micro_batch_size` to 1\n", + "3. **Still OOM:** Reduce `global_batch_size` and `max_seq_length`\n", + "4. **Last resort:** Increase `num_gpus_per_node`\n", + "\n", + "**No chat template / `/chat/completions` fails:**\n", + "- Use Instruct model variants (e.g., `Llama-3.2-1B-Instruct`) instead of base models (`Llama-3.2-1B`). Base models do not include a chat template in their tokenizer, so the output model will also lack one.\n", + "\n", + "**Distilled model quality is poor:**\n", + "- Increase `distillation_temperature` (try 2.0–5.0) to transfer more nuanced knowledge\n", + "- Adjust `distillation_ratio`—if dataset labels are high-quality, lower the ratio; if the teacher is strong, raise it\n", + "- Increase `epochs` or `max_steps` for more training\n", + "- Verify teacher and student share the same vocabulary\n", + "\n", + "**Vocabulary mismatch error:**\n", + "- Teacher and student must use the same tokenizer. Use models from the same family (e.g., Llama 3.2 1B Instruct + Llama 3.2 3B Instruct)\n", + "\n", + "**Deployment fails:**\n", + "- Verify output model exists: `client.models.retrieve(name=DISTILLED_STUDENT_NAME, workspace=\"default\")`\n", + "- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n", + "- The distilled model has the same size as the student, so GPU requirements match the student model\n", + "\n", + "\n", + "## Next Steps\n", + "\n", + "- [Monitor training metrics](fine-tune-metrics) in detail\n", + "- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n", + "- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning\n", + "- Learn about [Full SFT](./sft-customization-job) for direct supervised fine-tuning" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": ".venv", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.11.14" + } + }, + "nbformat": 4, + "nbformat_minor": 4 } diff --git a/docs/customizer/tutorials/distillation-customization-job.mdx b/docs/customizer/tutorials/distillation-customization-job.mdx index 3553b95f6b..8da102ae5f 100644 --- a/docs/customizer/tutorials/distillation-customization-job.mdx +++ b/docs/customizer/tutorials/distillation-customization-job.mdx @@ -3,7 +3,730 @@ title: "Knowledge Distillation Customization" description: "" --- - +[Run in Google Colab](https://colab.research.google.com/github/NVIDIA-NeMo/nemo-platform/blob/main/docs/customizer/tutorials/distillation-customization-job.ipynb) + +# Knowledge Distillation Customization + +Learn how to train a smaller student model to mimic a larger teacher model using knowledge distillation (KD). + +## About + +Knowledge Distillation transfers knowledge from a large **teacher** model to a smaller **student** model. During training, the student learns to match the teacher's output probability distribution, producing a compact model that retains much of the teacher's capability. + +**What you can achieve with KD:** + +- **Compress models:** Distill a 3B model into a 1B model for faster inference and lower deployment costs +- **Reduce latency:** Deploy a smaller model that responds faster while preserving quality +- **Lower resource requirements:** Serve a distilled model on fewer GPUs + +### KD vs SFT: Understanding the Trade-offs + +| Aspect | Full SFT | Knowledge Distillation | +| --- | --- | --- | +| **Training signal** | Ground-truth labels only | Teacher's soft probability distribution + labels | +| **Knowledge source** | Dataset examples | Teacher model's learned representations | +| **Output model size** | Same as input model | Typically a smaller student model | +| **GPU requirements** | Needs to fit one model | Needs to fit both teacher and student in memory | +| **Best for** | Domain adaptation, new knowledge injection | Model compression, latency reduction | + +### Key Parameters + +| Parameter | Default | Description | +| --- | --- | --- | +| `teacher_model` | *(required)* | Teacher model entity URN (e.g., `default/llama-3-2-3b-teacher`) | +| `teacher_precision` | `bf16` | Precision for the frozen teacher (`bf16`, `fp16`, `fp32`). Lower = less memory | +| `distillation_ratio` | `0.5` | Balance between CE loss and KD loss. `0.0` = CE only, `1.0` = KD only | +| `distillation_temperature` | `1.0` | Softmax temperature. Higher = softer distributions, more knowledge transfer | + +### Workflow Overview + +This tutorial follows a complete distillation pipeline: + +1. **Fine-tune the teacher** (SFT on the task dataset) so it learns the domain +2. **Establish a baseline** by deploying the base student model and measuring ROUGE scores +3. **Distill into the student** using the fine-tuned teacher's soft targets +4. **Evaluate the distilled student** and compare ROUGE scores against the baseline + +**When to choose KD:** + +- You have a high-quality large model and want a smaller, faster version +- Deployment latency or cost is a constraint +- The teacher and student share the same vocabulary (e.g., both are Llama models) + +**When to choose SFT instead:** Refer to the [Full SFT tutorial](/documentation/customizer-reference/tutorials/sft-customization-job) when you want to train a model directly on labeled data without a teacher. + +## Prerequisites + +Before starting this tutorial, ensure you have: + +1. **Completed the [Quickstart](/documentation/get-started)** to install and deploy NeMo Platform locally +2. **Installed the Python SDK** (PyPI wrapper: `pip install "nemo-platform[all]"`; source checkout: run `make bootstrap` from the repository root) +3. **Installed evaluation dependencies:** + +```sh +pip install evaluate rouge_score datasets +``` + +## Quick Start + +### 1. Initialize SDK + +The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](/documentation/get-started) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable: + +```sh +export NMP_BASE_URL= +``` + +```python +import json +import os +import time +import uuid +from pathlib import Path + +from nemo_platform import NeMoPlatform, ConflictError + +NMP_BASE_URL = os.environ.get("NMP_BASE_URL", "http://localhost:8080") +client = NeMoPlatform( + base_url=NMP_BASE_URL, + workspace="default" +) +``` + +### 2. Prepare Dataset + +Knowledge distillation uses the same dataset formats as SFT. We use the SQuAD dataset for both teacher training and distillation so that the teacher first learns the task, then transfers that knowledge to the student. + +We also hold out a small **test split** for ROUGE evaluation at the end. + +```python +from datasets import load_dataset, DatasetDict + +print("Loading dataset rajpurkar/squad") +raw_dataset = load_dataset("rajpurkar/squad") +if not isinstance(raw_dataset, DatasetDict): + raise ValueError("Dataset does not contain expected splits") + +print("Loaded dataset") + +SEED = 1234 +TRAINING_SIZE = 3000 +VALIDATION_SIZE = 300 +TEST_SIZE = 100 +DATASET_PATH = Path("kd-dataset").absolute() + +os.makedirs(DATASET_PATH, exist_ok=True) + +train_set = raw_dataset.get('train') +split = train_set.train_test_split(test_size=0.05, seed=SEED) + +train_ds = split['train'].select(range(min(TRAINING_SIZE, len(split['train'])))) +val_ds = split['test'].select(range(min(VALIDATION_SIZE, len(split['test'])))) +test_ds = split['test'].select(range(VALIDATION_SIZE, min(VALIDATION_SIZE + TEST_SIZE, len(split['test'])))) + + +def convert_squad(example): + """Convert SQuAD format to prompt/completion format.""" + prompt = f"Context: {example['context']} Question: {example['question']} Answer:" + completion = example["answers"]["text"][0] + return {"prompt": prompt, "completion": completion} + + +def write_jsonl(dataset, path): + with open(path, "w", encoding="utf-8") as f: + for example in dataset: + f.write(json.dumps(convert_squad(example)) + "\n") + + +def write_test_jsonl(dataset, path): + """Save test split with raw context/question for chat-style evaluation.""" + with open(path, "w", encoding="utf-8") as f: + for example in dataset: + f.write(json.dumps({ + "context": example["context"], + "question": example["question"], + "completion": example["answers"]["text"][0], + }) + "\n") + + +write_jsonl(train_ds, f"{DATASET_PATH}/training.jsonl") +write_jsonl(val_ds, f"{DATASET_PATH}/validation.jsonl") +write_test_jsonl(test_ds, f"{DATASET_PATH}/testing.jsonl") + +print(f"Training: {len(train_ds)} rows") +print(f"Validation: {len(val_ds)} rows") +print(f"Test: {len(test_ds)} rows") + +with open(f"{DATASET_PATH}/training.jsonl", 'r') as f: + sample = json.loads(f.readline()) + print(f"\nSample prompt: {sample['prompt'][:150]}...") + print(f"Sample completion: {sample['completion']}") +``` + +```python +DATASET_NAME = "kd-dataset" + +try: + client.files.filesets.create( + workspace="default", + name=DATASET_NAME, + description="Knowledge distillation training data" + ) + print(f"Created fileset: {DATASET_NAME}") +except ConflictError: + print(f"Fileset '{DATASET_NAME}' already exists, continuing...") + +client.files.upload( + local_path=f"{DATASET_PATH}/", + remote_path="", + fileset=DATASET_NAME, + workspace="default" +) + +print("Uploaded files:") +print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2)) +``` + +### 3. Secrets Setup + +In this tutorial we use two Llama 3.2 Instruct models from HuggingFace: +- **Teacher:** [meta-llama/Llama-3.2-3B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct) (3B parameters) +- **Student:** [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) (1B parameters) + +Both models share the same tokenizer/vocabulary (required for knowledge distillation) and include a chat template for deployment with `/chat/completions`. + +**HuggingFace Authentication:** +- For gated models (Llama, Gemma), you must provide a HuggingFace token via the `token_secret` parameter +- Get your token from [HuggingFace Settings](https://huggingface.co/settings/tokens) (requires Read access) +- Accept the model's terms on the HuggingFace model page before using it: + - [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) + - [meta-llama/Llama-3.2-3B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct) + +```python +HF_TOKEN = os.getenv("HF_TOKEN") + + +def create_or_get_secret(name: str, value: str | None, label: str): + if not value: + raise ValueError(f"{label} is not set") + try: + secret = client.secrets.create( + name=name, + workspace="default", + value=value, + ) + print(f"Created secret: {name}") + return secret + except ConflictError: + print(f"Secret '{name}' already exists, continuing...") + return client.secrets.retrieve(name=name, workspace="default") + + +hf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN") +print("HF_TOKEN secret:") +print(hf_secret.model_dump_json(indent=2)) +``` + +### 4. Create Model FileSets and Model Entities + +Knowledge distillation requires **two** model entities: +1. **Student model** — the smaller model that will be trained ([meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct)) +2. **Teacher model** — the larger model that provides soft targets ([meta-llama/Llama-3.2-3B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct)) + +Using the Instruct variants ensures the output model includes a chat template, which is required for the `/chat/completions` inference endpoint. + +```python +from nemo_platform.types.files import HuggingfaceStorageConfigParam + +SPEC_TIMEOUT_SECONDS = 120 + + +def create_model(hf_repo: str, model_name: str, description: str): + """Create a fileset + model entity and wait for ModelSpec.""" + try: + client.files.filesets.create( + workspace="default", + name=model_name, + description=description, + storage=HuggingfaceStorageConfigParam( + type="huggingface", + repo_id=hf_repo, + repo_type="model", + token_secret=hf_secret.name + ) + ) + print(f"Created fileset: {model_name}") + except ConflictError: + print(f"Fileset '{model_name}' already exists.") + + try: + model = client.models.create( + workspace="default", + name=model_name, + fileset=f"default/{model_name}", + ) + print(f"Created Model Entity: {model_name}") + except ConflictError: + print(f"Model '{model_name}' already exists. Updating fileset.") + model = client.models.update( + workspace="default", + name=model_name, + fileset=f"default/{model_name}", + ) + + print(f"Waiting for ModelSpec on {model_name}...") + spec_start = time.time() + while not model.spec: + if time.time() - spec_start > SPEC_TIMEOUT_SECONDS: + raise TimeoutError(f"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS}s") + time.sleep(2) + model = client.models.retrieve(workspace="default", name=model_name) + print(f"ModelSpec populated: {model.spec}") + return model + + +student_model = create_model( + hf_repo="meta-llama/Llama-3.2-1B-Instruct", + model_name="llama-3-2-1b-student", + description="Llama 3.2 1B Instruct student model", +) + +print() + +teacher_model = create_model( + hf_repo="meta-llama/Llama-3.2-3B-Instruct", + model_name="llama-3-2-3b-teacher", + description="Llama 3.2 3B Instruct teacher model", +) +``` + +--- + +## Phase 1: Fine-Tune the Teacher + +### 5. Train Teacher with Full SFT + +For best distillation results, fine-tune the teacher on the **same dataset** that will be used for distillation. This ensures the teacher has learned the task-specific knowledge that the student will inherit. + +We train the 3B Instruct model with Full SFT on the SQuAD dataset. + +```python +from nemo_automodel_plugin.schema import AutomodelJobInput + +job_suffix = uuid.uuid4().hex[:4] + +TEACHER_JOB_NAME = f"teacher-sft-job-{job_suffix}" +TEACHER_OUTPUT_NAME = f"teacher-model-{job_suffix}" + +teacher_spec = AutomodelJobInput( + model=f"default/{teacher_model.name}", + dataset={"training": f"default/{DATASET_NAME}"}, + training={ + "training_type": "sft", + "finetuning_type": "all_weights", + "max_seq_length": 2048, + }, + schedule={"epochs": 1}, + batch={"global_batch_size": 64, "micro_batch_size": 1}, + optimizer={"learning_rate": 5e-5}, + parallelism={"num_gpus_per_node": 1}, + output={"name": TEACHER_OUTPUT_NAME}, +) + +teacher_job = client.customization.automodel.jobs.create( + spec=teacher_spec, workspace="default", name=TEACHER_JOB_NAME +) + +TRAINED_TEACHER_NAME = TEACHER_OUTPUT_NAME +print(f"Teacher training job: {teacher_job.job.name}") +print(f"Output teacher model: {TRAINED_TEACHER_NAME}") + +``` + +```python +from IPython.display import clear_output + + +def wait_for_job(job_name: str): + """Poll job status until completion.""" + while True: + status = client.jobs.get_status(name=job_name, workspace="default") + clear_output(wait=True) + print(f"Job: {job_name}") + print(f"Status: {status.status}") + + for job_step in status.steps or []: + if job_step.name == "training": + for task in job_step.tasks or []: + details = task.status_details or {} + step = details.get("step") + max_steps = details.get("max_steps") + if step is not None and max_steps is not None: + print(f"Progress: Step {step}/{max_steps} ({step / max_steps * 100:.1f}%)") + phase = details.get("phase") + if phase: + print(f"Phase: {phase}") + break + break + + if status.status in ("completed", "failed", "cancelled", "error"): + print(f"\nJob finished: {status.status}") + return status + + time.sleep(10) + + +teacher_status = wait_for_job(TEACHER_JOB_NAME) +assert teacher_status.status == "completed" +``` + +--- + +## Phase 2: Establish Baseline (Base Student) + +### 6. Deploy the Base Student Model + +Before distillation, deploy the base student model (1B Instruct, without any fine-tuning) to establish a baseline ROUGE score. After distillation, we compare the distilled student against this baseline to measure improvement. + +```python +baseline_suffix = uuid.uuid4().hex[:4] +BASELINE_DEPLOYMENT_CONFIG = f"baseline-student-cfg-{baseline_suffix}" +BASELINE_DEPLOYMENT_NAME = f"baseline-student-{baseline_suffix}" + +baseline_deployment_config = client.inference.deployment_configs.create( + workspace="default", + name=BASELINE_DEPLOYMENT_CONFIG, + engine="vllm", + model_spec={ + "model_namespace": "default", + "model_name": student_model.name, + }, + executor_config={ + "gpu": 1, + "image_name": "vllm/vllm-openai", + "image_tag": "v0.22.1", + }, +) + +baseline_deployment = client.inference.deployments.create( + workspace="default", + name=BASELINE_DEPLOYMENT_NAME, + config=baseline_deployment_config.name +) + +print(f"Baseline student deployment: {baseline_deployment.name}") + +``` + +```python +def wait_for_deployment(deployment_name: str, timeout_minutes: int = 30): + """Poll deployment until ready.""" + start = time.time() + timeout = timeout_minutes * 60 + while True: + dep = client.inference.deployments.retrieve(name=deployment_name, workspace="default") + elapsed = time.time() - start + clear_output(wait=True) + print(f"Deployment: {deployment_name}") + print(f"Status: {dep.status}") + print(f"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s") + + if dep.status == "READY": + print("\nDeployment is ready!") + return dep + if dep.status in ("FAILED", "ERROR", "TERMINATED", "LOST"): + print(f"\nDeployment failed: {dep.status}") + return dep + if elapsed > timeout: + print(f"\nTimeout ({timeout_minutes}m). Check status manually.") + return dep + time.sleep(15) + + +dep_status = wait_for_deployment(BASELINE_DEPLOYMENT_NAME) +assert dep_status.status == "READY" +``` + +### 7. Generate Baseline Predictions on Test Set + +```python +with open(f"{DATASET_PATH}/testing.jsonl", "r", encoding="utf-8") as f: + test_data = [json.loads(line) for line in f] + +contexts = [row["context"] for row in test_data] +questions = [row["question"] for row in test_data] +reference_completions = [row["completion"] for row in test_data] + +print(f"Test samples: {len(contexts)}") +print(f"Sample context: {contexts[0]}") +print(f"Sample question: {questions[0]}") +print(f"Sample reference: {reference_completions[0]}") +``` + +```python +def generate_completions( + deployment_name: str, + output_model_name: str, + contexts: list[str], + questions: list[str], +) -> list[str]: + """Generate completions for a list of context/question pairs using a deployed model.""" + completions = [] + for context, question in zip(contexts, questions): + messages = [ + { + "role": "user", + "content": f"Based on the following context, answer the question.\n\nContext: {context}\n\nQuestion: {question}", + } + ] + response = client.inference.gateway.provider.post( + "v1/chat/completions", + name=deployment_name, + workspace="default", + body={ + "model": f"default/{output_model_name}", + "messages": messages, + "temperature": 0, + "max_tokens": 128, + } + ) + completions.append(response["choices"][0]["message"]["content"]) + return completions + + +print("Generating baseline (base student) predictions...") +baseline_completions = generate_completions(BASELINE_DEPLOYMENT_NAME, student_model.name, contexts, questions) +print(f"Generated {len(baseline_completions)} baseline predictions") +print(f"\nSample baseline output: {baseline_completions[0]}") +``` + +### 8. Delete Baseline Deployment + +Delete the baseline student deployment to free GPU resources for the distillation training job and subsequent distilled model deployment. + +```python +client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace="default") +print(f"Deleted baseline deployment: {BASELINE_DEPLOYMENT_NAME}") + +if not client.models.wait_for_status( + deployment_name=BASELINE_DEPLOYMENT_NAME, + desired_status="DELETED", + workspace="default", + timeout=600, +): + raise TimeoutError( + f"Deployment {BASELINE_DEPLOYMENT_NAME} was not deleted within timeout" + ) + +client.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace="default") +print(f"Deleted baseline deployment config: {BASELINE_DEPLOYMENT_CONFIG}") +``` + +--- + +## Phase 3: Distill into Student + +### 9. Create Knowledge Distillation Job + +Now create a distillation job that trains the 1B student using the **fine-tuned** 3B teacher's output distribution. The `model` field specifies the student, and `teacher_model` references the trained teacher model entity from Phase 1. + +**GPU Requirements:** + +KD requires loading both student and teacher models, so plan GPU memory accordingly: +- 1B student + 3B teacher: 1 GPU (24GB+ VRAM each) +- 3B student + 8B teacher: 4 GPUs +- 8B student + 70B teacher: 8+ GPUs + +Use `teacher_precision="bf16"` (default) to reduce teacher memory footprint. + +```python +from nemo_automodel_plugin.schema import AutomodelJobInput + +KD_JOB_NAME = f"my-kd-job-{job_suffix}" +KD_OUTPUT_NAME = f"kd-student-{job_suffix}" + +kd_spec = AutomodelJobInput( + model=f"default/{student_model.name}", + dataset={"training": f"default/{DATASET_NAME}"}, + training={ + "training_type": "distillation", + "finetuning_type": "all_weights", + "teacher_model": f"default/{TRAINED_TEACHER_NAME}", + "teacher_precision": "bf16", + "distillation_ratio": 0.5, + "distillation_temperature": 2.0, + "max_seq_length": 2048, + }, + schedule={"epochs": 1}, + batch={"global_batch_size": 64, "micro_batch_size": 1}, + optimizer={"learning_rate": 5e-5}, + parallelism={"num_gpus_per_node": 1}, + output={"name": KD_OUTPUT_NAME}, +) + +kd_job = client.customization.automodel.jobs.create( + spec=kd_spec, workspace="default", name=KD_JOB_NAME +) + +DISTILLED_STUDENT_NAME = KD_OUTPUT_NAME +print(f"Distillation job: {kd_job.job.name}") +print(f"Output student model: {DISTILLED_STUDENT_NAME}") + +``` + +### 10. Track Distillation Progress + +```python +kd_status = wait_for_job(KD_JOB_NAME) +assert kd_status.status == "completed" +``` + +--- + +## Phase 4: Evaluate the Distilled Student Model + +### 11. Deploy the Distilled Student Model + +The output model has the same architecture as the 1B student—only its weights have been updated via distillation. It requires just 1 GPU to deploy. + +```python +deploy_suffix_2 = uuid.uuid4().hex[:4] +STUDENT_DEPLOYMENT_CONFIG = f"kd-student-deploy-cfg-{deploy_suffix_2}" +STUDENT_DEPLOYMENT_NAME = f"kd-student-deploy-{deploy_suffix_2}" + +student_deployment_config = client.inference.deployment_configs.create( + workspace="default", + name=STUDENT_DEPLOYMENT_CONFIG, + engine="vllm", + model_spec={ + "model_namespace": "default", + "model_name": DISTILLED_STUDENT_NAME, + }, + executor_config={ + "gpu": 1, + "image_name": "vllm/vllm-openai", + "image_tag": "v0.22.1", + }, +) + +student_deployment = client.inference.deployments.create( + workspace="default", + name=STUDENT_DEPLOYMENT_NAME, + config=student_deployment_config.name +) + +print(f"Student deployment: {student_deployment.name}") + +``` + +```python +wait_for_deployment(STUDENT_DEPLOYMENT_NAME) +``` + +### 12. Generate Student Predictions on Test Set + +```python +print("Generating distilled student predictions...") +student_completions = generate_completions(STUDENT_DEPLOYMENT_NAME, DISTILLED_STUDENT_NAME, contexts, questions) +print(f"Generated {len(student_completions)} student predictions") +print(f"\nSample student output: {student_completions[0]}") +``` + +### 13. Compute ROUGE Scores + +Compare the base student (before distillation) and the distilled student against the ground-truth reference completions using ROUGE metrics. + +```python +import evaluate + +rouge = evaluate.load("rouge") + +baseline_scores = rouge.compute(predictions=baseline_completions, references=reference_completions) +student_scores = rouge.compute(predictions=student_completions, references=reference_completions) + +metrics = list(baseline_scores.keys()) +header = f"{'Model':<35} " + " ".join(f"{m:>10}" for m in metrics) +separator = "-" * len(header) + +print("=" * 60) +print("ROUGE SCORE COMPARISON") +print("=" * 60) +print(header) +print(separator) +print(f"{'Base Student (1B, no training)':<35} " + " ".join(f"{baseline_scores[m]:>10.4f}" for m in metrics)) +print(f"{'Distilled Student (1B, KD)':<35} " + " ".join(f"{student_scores[m]:>10.4f}" for m in metrics)) +``` + +```python +print("Sample predictions (first 3):\n") +for i in range(min(3, len(contexts))): + print(f"--- Sample {i + 1} ---") + print(f"Question: {questions[i]}") + print(f"Reference: {reference_completions[i]}") + print(f"Baseline: {baseline_completions[i][:200]}") + print(f"Distilled: {student_completions[i][:200]}") + print() +``` + +**Interpreting ROUGE Scores:** + +| Metric | Measures | +|--------|----------| +| **ROUGE-1** | Unigram overlap between prediction and reference | +| **ROUGE-2** | Bigram overlap (captures phrase-level similarity) | +| **ROUGE-L** | Longest common subsequence (captures sentence structure) | +| **ROUGE-Lsum** | ROUGE-L computed over full summaries | + +**What to expect:** +- The base student (1B, no training) provides a lower bound since it has not seen the task data +- The distilled student (1B, KD) should significantly outperform the base student, demonstrating the knowledge transferred from the 3B teacher +- If the distilled student scores are not much higher than the baseline, try increasing `distillation_temperature`, adjusting `distillation_ratio`, or training for more epochs + +--- + +## Hyperparameters + +For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](/documentation/customizer-reference/manage-customization-jobs/training-configuration). + +--- + +## Troubleshooting + +**Job fails during model download:** +- Verify authentication secrets are configured (refer to [Managing Secrets](/documentation/get-started/core-concepts/manage-secrets)) +- For gated HuggingFace models (Llama, Gemma), accept the license on the model page +- Check both `model` (student) and `teacher_model` URNs are correct +- Ensure both model entities exist: `client.models.retrieve(name=..., workspace="default")` + +**Job fails with OOM (Out of Memory) error:** + +KD loads both models, so OOM is more likely than with SFT: +1. **First try:** Use `teacher_precision="bf16"` to reduce teacher memory +2. **Still OOM:** Reduce `micro_batch_size` to 1 +3. **Still OOM:** Reduce `global_batch_size` and `max_seq_length` +4. **Last resort:** Increase `num_gpus_per_node` + +**No chat template / `/chat/completions` fails:** +- Use Instruct model variants (e.g., `Llama-3.2-1B-Instruct`) instead of base models (`Llama-3.2-1B`). Base models do not include a chat template in their tokenizer, so the output model will also lack one. + +**Distilled model quality is poor:** +- Increase `distillation_temperature` (try 2.0–5.0) to transfer more nuanced knowledge +- Adjust `distillation_ratio`—if dataset labels are high-quality, lower the ratio; if the teacher is strong, raise it +- Increase `epochs` or `max_steps` for more training +- Verify teacher and student share the same vocabulary + +**Vocabulary mismatch error:** +- Teacher and student must use the same tokenizer. Use models from the same family (e.g., Llama 3.2 1B Instruct + Llama 3.2 3B Instruct) + +**Deployment fails:** +- Verify output model exists: `client.models.retrieve(name=DISTILLED_STUDENT_NAME, workspace="default")` +- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace="default")` +- The distilled model has the same size as the student, so GPU requirements match the student model + + +## Next Steps + +- [Monitor training metrics](/documentation/customizer-reference/tutorials/metrics) in detail +- [Evaluate your fine-tuned model](/documentation/evaluate-models) using the Evaluator service +- Learn about [LoRA customization](/documentation/customizer-reference/tutorials/lora-customization-job) for resource-efficient fine-tuning +- Learn about [Full SFT](/documentation/customizer-reference/tutorials/sft-customization-job) for direct supervised fine-tuning diff --git a/docs/customizer/tutorials/dpo-customization-job.ipynb b/docs/customizer/tutorials/dpo-customization-job.ipynb deleted file mode 100644 index 0e854be47f..0000000000 --- a/docs/customizer/tutorials/dpo-customization-job.ipynb +++ /dev/null @@ -1,963 +0,0 @@ -{ - "cells": [ - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "\n", - "\n", - "\n", - "# DPO Customization\n", - "\n", - "Learn how to use the NeMo Platform to create a DPO (Direct Preference Optimization) job using a custom dataset.\n", - "\n", - "## About\n", - "\n", - "DPO is an advanced fine-tuning technique for preference-based alignment. If you're new to fine-tuning, consider starting with [LoRA](./lora-customization-job) or [Full SFT](./sft-customization-job) tutorials first.\n", - "\n", - "Direct Preference Optimization (DPO) is an RL-free alignment algorithm that operates on preference data. Given a prompt and a pair of chosen and rejected responses, DPO aims to increase the probability of the chosen response and decrease the probability of the rejected response relative to a frozen reference model. The actor is initialized using the reference model. For more details, refer to the [DPO paper](https://arxiv.org/pdf/2305.18290).\n", - "\n", - "DPO shares similarities with Full SFT training workflows but differs in a few key ways:\n", - "\n", - "| Aspect | SFT (Supervised Fine-Tuning) | DPO (Direct Preference Optimization) |\n", - "| --- | --- | --- |\n", - "| Data Requirements | Labeled instruction-response pairs where the desired output is explicitly provided | Pairwise preference data, where for a given input, one response is explicitly preferred over another |\n", - "| Learning Objective | Directly teaches the model to generate a specific \"correct\" response | Directly optimizes the model to align with human preferences by maximizing the probability of preferred responses and minimizing rejected ones, without needing an explicit reward model |\n", - "| Alignment Focus | Aligns the model with the specific examples present in its training data | Aligns the model with broader human preferences, which can be more effective for subjective tasks or those without a single \"correct\" answer |\n", - "| Computational Efficiency | Standard fine-tuning efficiency | More computationally efficient than SFT (especially when compared to full RLHF methods) as it bypasses the need to train a separate reward model |\n", - "\n", - "**What you can achieve with DPO:**\n", - "- **Align with human preferences**: Directly optimize your model to produce outputs that align with subjective human preferences without requiring explicit reward modeling\n", - "- **Refine response quality**: Improve helpfulness, harmlessness, honesty, and other nuanced qualities that are easier to compare than to define\n", - "- **Control tone and style**: Adjust the model's communication style, verbosity, formality, and other subjective characteristics\n", - "- **Implement safety guardrails**: Teach the model to avoid harmful or undesirable responses by training on preferred vs. rejected response pairs\n", - "- **Optimize subjective tasks**: Excel at tasks where there are multiple acceptable answers but clear preferences exist (creative writing, dialogue, explanations)\n", - "\n", - "**When to choose DPO:**\n", - "- **Subjective quality matters**: Your task involves style, tone, or other qualities where there's no single \"correct\" answer but clear preferences exist\n", - "- **You have preference data**: You can collect pairwise comparisons (preferred vs. rejected responses) more easily than perfect labeled examples\n", - "- **Refining existing capabilities**: You want to make targeted improvements to an already-trained model without major capability changes\n", - "- **Complex evaluation**: Humans find it easier to compare which of two responses is better than to create the ideal response themselves (especially for multi-turn conversations, creative tasks, or nuanced outputs)\n", - "- **Robust behavior changes**: You need more reliable behavior modification than prompting can provide, without the complexity of full RLHF\n", - "- **Lower compute than RLHF**: You want human preference alignment but with simpler training that doesn't require reinforcement learning infrastructure\n", - "\n", - "**When to choose SFT:**\n", - "- **Clear correct answers**: Your task has objectively correct outputs (code generation, structured data extraction, following specific formats)\n", - "- **High-quality examples**: You have well-labeled input-output pairs that demonstrate exactly what the model should produce\n", - "- **Imitation learning**: You want the model to closely mimic a specific style, format, or knowledge base from expert demonstrations\n", - "- **Foundational capabilities**: You're establishing new task-specific capabilities before fine-tuning preferences (SFT is often done before DPO)\n", - "- **Stable, predictable outputs**: You need consistent formatting or structure that's well-defined in your training examples\n", - "- **Traditional NLP tasks**: Instruction following, translation, summarization, or classification where gold-standard labels exist" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Prerequisites\n", - "\n", - "Before starting this tutorial, ensure you have:\n", - "\n", - "1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n", - "2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Quick Start\n", - "\n", - "### 1. Initialize SDK\n", - "\n", - "The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](../../get-started/quickstart.md) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable:\n", - "\n", - "```sh\n", - "export NMP_BASE_URL=\n", - "```" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import json\n", - "import os\n", - "from nemo_platform import NeMoPlatform, ConflictError\n", - "from nemo_platform.types.customization import (\n", - " CustomizationJobInputParam,\n", - " DpoTrainingParam,\n", - " ParallelismParamsParam,\n", - ")\n", - "\n", - "NMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\n", - "client = NeMoPlatform(\n", - " base_url=NMP_BASE_URL,\n", - " workspace=\"default\"\n", - ")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 2. Prepare Dataset\n", - "\n", - "Create your data in JSONL format - one JSON object per line. The platform auto-detects your data format. Supported dataset formats are listed below.\n", - "\n", - "**Flexible Data Setup:**\n", - "- **No validation file?** The platform automatically creates a 10% validation split\n", - "- **Multiple files?** Upload to `training/` or `validation/` subdirectories—they'll be automatically merged\n", - "- **Format detection:** Your data format is auto-detected at training time\n", - "\n", - "In this tutorial the following dataset directory structure will be used:\n", - "```\n", - "my_dataset\n", - "`-- training.jsonl\n", - "`-- validation.jsonl\n", - "```" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### Binary Preference Format\n", - "DPO training requires preference pairs with three fields:\n", - "- **`prompt`**: The input prompt (can be a string or array of message objects)\n", - "- **`chosen`**: The preferred response\n", - "- **`rejected`**: The less preferred response" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "language": "json" - }, - "outputs": [], - "source": [ - "{\"prompt\": [{\"role\": \"user\", \"content\": \"What is the capital of France?\"}], \"chosen\": \"The capital of France is Paris. It is the largest city in France and serves as the country's political, economic, and cultural center.\", \"rejected\": \"I think the capital of France might be London or Paris, I'm not entirely sure.\"}" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### Tulu3 Preference Dataset Format\n", - "This format contains complete conversation histories for both the chosen (preferred) and rejected responses.\n", - "\n", - "Required fields:\n", - "- **`chosen`**: Full conversation with the preferred response (list of message objects, last must be assistant)\n", - "- **`rejected`**: Full conversation with the rejected response (list of message objects, last must be assistant)" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "{\"chosen\": [{\"role\": \"user\", \"content\": \"What is the capital of France?\"}, {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"}], \"rejected\": [{\"role\": \"user\", \"content\": \"What is the capital of France?\"}, {\"role\": \"assistant\", \"content\": \"I'm not sure, but I think it might be London or Paris.\"}]}" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### HelpSteer Dataset Format\n", - "This format uses numeric preference scores to indicate which response is better. The context can be either a simple string or an array of message objects.\n", - "\n", - "Required fields:\n", - "- **`context`**: The input context (can be a string or array of message objects)\n", - "- **`response1`**: First response option\n", - "- **`response2`**: Second response option\n", - "- **`overall_preference`**: Preference score where negative values mean response1 is preferred, positive values mean response2 is preferred, and 0 indicates a tie" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "language": "json" - }, - "outputs": [], - "source": [ - "{\"context\": \"Explain how to use git rebase\", \"response1\": \"Git rebase is a command that rewrites commit history by moving or combining commits. Use 'git rebase main' to reapply your branch commits on top of main. This creates a linear history and avoids merge commits.\", \"response2\": \"Use git rebase to change commits. Just type git rebase and it will work.\", \"overall_preference\": -2}" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 3. Create Dataset FileSet and Upload Training Data" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "Install huggingface datasets package to download public [nvidia/HelpSteer3](https://huggingface.co/datasets/nvidia/HelpSteer3) dataset if it's not installed in your Python environment:\n", - "\n", - "```sh\n", - "pip install datasets\n", - "```" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### Download nvidia/HelpSteer3 Dataset" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from pathlib import Path\n", - "from datasets import load_dataset, Dataset\n", - "ds = load_dataset(\"nvidia/HelpSteer3\", \"preference\")\n", - "\n", - "# Adjust these values to change the size of the training and validation sets\n", - "# The larger the datasets, the better the model will perform but longer the training will take\n", - "# For the purpose of this tutorial, we'll use a small subset of the dataset\n", - "training_size = 3000\n", - "validation_size = 300\n", - "DATASET_PATH = Path(\"dpo-dataset\").absolute()\n", - "\n", - "# Get training split and verify it's a Dataset (not IterableDataset)\n", - "train_dataset = ds[\"train\"]\n", - "validation_dataset = ds[\"validation\"]\n", - "assert isinstance(train_dataset, Dataset), \"Expected Dataset type\"\n", - "assert isinstance(validation_dataset, Dataset), \"Expected Dataset type\"\n", - "\n", - "# Select subsets and save to JSONL files\n", - "testing_ds = train_dataset.select(range(training_size))\n", - "validation_ds = validation_dataset.select(range(validation_size))\n", - "\n", - "# Create directory if it doesn't exist\n", - "os.makedirs(DATASET_PATH, exist_ok=True)\n", - "\n", - "# Save subsets to JSONL files\n", - "testing_ds.to_json(f\"{DATASET_PATH}/training.jsonl\")\n", - "validation_ds.to_json(f\"{DATASET_PATH}/validation.jsonl\")\n", - "\n", - "print(f\"Saved training.jsonl with {len(testing_ds)} rows\")\n", - "print(f\"Saved validation.jsonl with {len(validation_ds)} rows\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### Upload Training Data" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Create fileset to store DPO training data\n", - "DATASET_NAME = \"dpo-dataset\"\n", - "\n", - "try:\n", - " client.files.filesets.create(\n", - " workspace=\"default\",\n", - " name=DATASET_NAME,\n", - " description=\"dpo training data\"\n", - " )\n", - " print(f\"Created fileset: {DATASET_NAME}\")\n", - "except ConflictError:\n", - " print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n", - "\n", - "# Upload training data files individually to ensure correct structure\n", - "client.files.upload(\n", - " local_path=DATASET_PATH, # Local directory with your JSONL files\n", - " remote_path=\"\",\n", - " fileset=DATASET_NAME,\n", - " workspace=\"default\"\n", - ")\n", - "\n", - "# Validate training data is uploaded correctly\n", - "print(\"Training data:\")\n", - "print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 4. Secrets Setup\n", - "\n", - "If you plan to use NGC or HuggingFace models, you'll need to configure authentication:\n", - "\n", - "- **NGC models** (`ngc://` URIs): Requires NGC API key\n", - "- **HuggingFace models** (`hf://` URIs): Requires HF token for gated/private models\n", - "\n", - "\n", - "Configure these as secrets in your platform. See [Managing Secrets](../../get-started/concepts/manage-secrets.md) for detailed instructions.\n", - "\n", - "Get your credentials to access base models:\n", - "- [NGC API Key](https://ngc.nvidia.com/) (Setup → Generate API Key)\n", - "- [HuggingFace Token](https://huggingface.co/settings/tokens) (Create token with Read access)\n", - "\n", - "\n", - "---\n", - "\n", - "#### Quick Setup Example\n", - "\n", - "In this tutorial we are going to work with [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model from HuggingFace. Ensure that you have sufficient permissions to download the model. If you cannot see the files in the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) Hugging Face page, request access\n", - "\n", - "**HuggingFace Authentication:**\n", - "- For gated models (Llama, Gemma), you must provide a HuggingFace token via the `token_secret` parameter\n", - "- Get your token from [HuggingFace Settings](https://huggingface.co/settings/tokens) (requires Read access)\n", - "- Accept the model's terms on the HuggingFace model page before using it. Example: [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main)\n", - "- For public models, you can omit the `token_secret` parameter when creating a fileset for model in the next step" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Export the HF_TOKEN and NGC_API_KEY environment variables if they are not already set\n", - "HF_TOKEN = os.getenv(\"HF_TOKEN\")\n", - "NGC_API_KEY = os.getenv(\"NGC_API_KEY\")\n", - "\n", - "\n", - "def create_or_get_secret(name: str, value: str | None, label: str):\n", - " if not value:\n", - " raise ValueError(f\"{label} is not set\")\n", - " try:\n", - " secret = client.secrets.create(\n", - " name=name,\n", - " workspace=\"default\",\n", - " value=value,\n", - " )\n", - " print(f\"Created secret: {name}\")\n", - " return secret\n", - " except ConflictError:\n", - " print(f\"Secret '{name}' already exists, continuing...\")\n", - " return client.secrets.retrieve(name=name, workspace=\"default\")\n", - "\n", - "\n", - "# Create HuggingFace token secret\n", - "hf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\n", - "print(\"HF_TOKEN secret:\")\n", - "print(hf_secret.model_dump_json(indent=2))\n", - "\n", - "# Create NGC API key secret\n", - "# Uncomment the line below if you have NGC API Key and want to finetune NGC models\n", - "# ngc_api_key = create_or_get_secret(\"ngc-api-key\", NGC_API_KEY, \"NGC_API_KEY\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 5. Create Base Model FileSet and Model Entity\n", - "\n", - "Create a fileset pointing to [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model in HuggingFace that we will train with DPO. Then create a Model Entity that references this fileset. Model downloading will take place at the DPO finetuning job creation time.\n", - "\n", - "Note: for public models, you can omit the `token_secret` parameter when creating a model fileset." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import time\n", - "\n", - "# Create a fileset pointing to the desired HuggingFace model\n", - "from nemo_platform.types.files import HuggingfaceStorageConfigParam\n", - "\n", - "HF_REPO_ID = \"meta-llama/Llama-3.2-1B-Instruct\"\n", - "MODEL_NAME = \"llama-3-2-1b-base\"\n", - "\n", - "# Ensure you have a HuggingFace token secret created\n", - "try:\n", - " base_model_fs = client.files.filesets.create(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " description=\"Llama 3.2 1B base model from HuggingFace\",\n", - " storage=HuggingfaceStorageConfigParam(\n", - " type=\"huggingface\",\n", - " # repo_id is the full model name from Hugging Face\n", - " repo_id=HF_REPO_ID,\n", - " repo_type=\"model\",\n", - " # we use the secret created in the previous step\n", - " token_secret=hf_secret.name\n", - " )\n", - " )\n", - " print(f\"Created base model fileset: {MODEL_NAME}\")\n", - "except ConflictError:\n", - " print(f\"Base model fileset already exists. Skipping creation.\")\n", - " base_model_fs = client.files.filesets.retrieve(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " )\n", - "\n", - "# Create Model Entity referencing the FileSet\n", - "try:\n", - " base_model = client.models.create(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " fileset=f\"default/{MODEL_NAME}\",\n", - " )\n", - " print(f\"Created Model Entity: {MODEL_NAME}\")\n", - "except ConflictError:\n", - " print(f\"Base model already exists. Updating fileset if different.\")\n", - " base_model = client.models.update(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " fileset=f\"default/{MODEL_NAME}\",\n", - " )\n", - "\n", - "print(f\"\\nBase model fileset: fileset://default/{base_model.name}\")\n", - "print(\"Base model fileset files list:\")\n", - "print(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=\"default\").data], indent=2))\n", - "\n", - "# Wait for ModelSpec to be populated from the checkpoint\n", - "print(\"\\nWaiting for ModelSpec to be populated...\")\n", - "SPEC_TIMEOUT_SECONDS = 120\n", - "spec_start = time.time()\n", - "while not base_model.spec:\n", - " if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n", - " raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n", - " time.sleep(2)\n", - " base_model = client.models.retrieve(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " )\n", - "\n", - "print(f\"ModelSpec populated: {base_model.spec}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 6. Create DPO Finetuning Job\n", - "Create a customization job with an inline target referencing the base model and dataset filesets created in previous steps." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "**Target `model_uri` Format:**\n", - "\n", - "Currently, `model_uri` must reference a FileSet:\n", - "- **FileSet:** `fileset://{workspace}/{fileset-name}`\n", - "\n", - "Support for direct HuggingFace (`hf://`) and NGC (`ngc://`) URIs is coming soon. For now, create a fileset as shown in the previous step, and the HuggingFace model will be downloaded at the beginning the finetuning job.\n", - "\n", - "**GPU Requirements:**\n", - "- 1B models: 1 GPU (24GB+ VRAM)\n", - "- 3B models: 1-2 GPUs \n", - "- 8B models: 2-4 GPUs\n", - "- 70B models: 8+ GPUs \n", - "\n", - "Adjust `num_gpus_per_node` and `tensor_parallel_size` based on your model size.\n", - "\n", - "**Important**\n", - "\n", - "When setting `val_check_interval` for DPO, use a fractional value (e.g., `0.5` for twice per epoch) or omit it entirely (validates once at end of epoch). Avoid integer step counts — they may not divide evenly into the total training steps, which can prevent validation from running on the final step." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import uuid\n", - "job_suffix = uuid.uuid4().hex[:4]\n", - "\n", - "JOB_NAME = f\"my-dpo-job-{job_suffix}\"\n", - "\n", - "job = client.customization.jobs.create(\n", - " name=JOB_NAME,\n", - " workspace=\"default\",\n", - " spec=CustomizationJobInputParam(\n", - " model=f\"default/{base_model.name}\",\n", - " dataset=f\"fileset://default/{DATASET_NAME}\",\n", - " training=DpoTrainingParam(\n", - " type=\"dpo\",\n", - " epochs=1,\n", - " batch_size=16,\n", - " learning_rate=0.00005,\n", - " max_seq_length=4096,\n", - " ref_policy_kl_penalty=0.1,\n", - " micro_batch_size=1,\n", - " parallelism=ParallelismParamsParam(\n", - " num_gpus_per_node=1,\n", - " num_nodes=1,\n", - " tensor_parallel_size=1,\n", - " pipeline_parallel_size=1,\n", - " ),\n", - " )\n", - " )\n", - ")\n", - "\n", - "print(f\"Job ID: {job.name}\")\n", - "print(f\"Output model: {job.spec.output.name}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 7. Track Training Progress" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import time\n", - "from IPython.display import clear_output\n", - "\n", - "# Poll job status every 10 seconds until completed\n", - "while True:\n", - " status = client.customization.jobs.get_status(\n", - " name=job.name,\n", - " workspace=\"default\"\n", - " )\n", - " \n", - " clear_output(wait=True)\n", - " print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n", - "\n", - " # Extract training progress from nested steps structure\n", - " step: int | None = None\n", - " max_steps: int | None = None\n", - " training_phase: str | None = None\n", - "\n", - " for job_step in status.steps or []:\n", - " if job_step.name == \"customization-training-job\":\n", - " for task in job_step.tasks or []:\n", - " task_details = task.status_details or {}\n", - " step = task_details.get(\"step\")\n", - " max_steps = task_details.get(\"max_steps\")\n", - " training_phase = task_details.get(\"phase\")\n", - " break\n", - " break\n", - "\n", - " if step is not None and max_steps is not None:\n", - " progress_pct = (step / max_steps) * 100\n", - " print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n", - " if training_phase:\n", - " print(f\"Training Phase: {training_phase}\")\n", - " else:\n", - " print(\"Training step not started yet or progress info not available\")\n", - " \n", - " # Exit loop when job is completed (or failed/cancelled)\n", - " if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n", - " print(f\"\\nJob finished with status: {status.status}\")\n", - " break\n", - " \n", - " time.sleep(10)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "**Interpreting DPO Training Metrics:**\n", - "\n", - "DPO training produces several key metrics:\n", - "\n", - "| Metric | Description | What to Look For |\n", - "|--------|-------------|------------------|\n", - "| **loss** | Total training loss (preference_loss + sft_loss) | Should decrease over training |\n", - "| **preference_loss** | Core DPO loss measuring preference learning | Starts near ln(2) ≈ 0.693, should decrease |\n", - "| **sft_loss** | SFT regularization term (often 0 for pure DPO) | Depends on configuration |\n", - "| **accuracy** | Fraction of samples where chosen > rejected | Should increase toward 80-95%+ |\n", - "| **rewards_chosen_mean** | Average implicit reward for chosen responses | Should be positive |\n", - "| **rewards_rejected_mean** | Average implicit reward for rejected responses | Should be negative |\n", - "\n", - "**Key Indicators:**\n", - "\n", - "- **Reward Margin** = `rewards_chosen_mean - rewards_rejected_mean`\n", - " - Should be positive and increasing\n", - " - Indicates the model is learning to distinguish preferences\n", - "\n", - "- **Accuracy Interpretation:**\n", - " - 50% = random chance (no learning)\n", - " - 66-75% = early/moderate learning\n", - " - 80%+ = good preference learning\n", - " - 95%+ = strong preference alignment\n", - "\n", - "**Troubleshooting:**\n", - "\n", - "- **Loss near ln(2) ≈ 0.693**: Model is at random chance level, training just starting or not learning\n", - "- **Accuracy stuck at ~50%**: Check data quality, increase learning rate, or verify preference labels\n", - "- **Negative reward margin**: Model is learning the wrong direction—check chosen/rejected labels\n", - "- **Loss increasing**: Learning rate too high or data quality issues\n", - "\n", - "**Note:** Training metrics measure optimization progress, not final model quality. Always evaluate the deployed model on your specific use case." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 8. Deploy Fine-Tuned Model\n", - "\n", - "Once training completes, deploy using the Deployment Management Service:" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Validate model entity exists\n", - "model_entity = client.models.retrieve(workspace='default', name=job.spec.output.name)\n", - "print(model_entity.model_dump_json(indent=2))" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from nemo_platform.types.inference import NIMDeploymentParam\n", - "\n", - "# Create deployment config\n", - "deploy_suffix = uuid.uuid4().hex[:4]\n", - "DEPLOYMENT_CONFIG_NAME = f\"dpo-model-deployment-cfg-{deploy_suffix}\"\n", - "DEPLOYMENT_NAME = f\"dpo-model-deployment-{deploy_suffix}\"\n", - "\n", - "deployment_config = client.inference.deployment_configs.create(\n", - " workspace=\"default\",\n", - " name=DEPLOYMENT_CONFIG_NAME,\n", - " nim_deployment=NIMDeploymentParam(\n", - " image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n", - " image_tag=\"1.13.1\",\n", - " gpu=1,\n", - " model_name=job.spec.output.name, # ModelEntity name from training,\n", - " model_namespace=\"default\", # Workspace where ModelEntity lives\n", - " )\n", - ")\n", - "\n", - "# Deploy model using deployment_config created above\n", - "deployment = client.inference.deployments.create(\n", - " workspace=\"default\",\n", - " name=DEPLOYMENT_NAME,\n", - " config=deployment_config.name\n", - ")\n", - "\n", - "\n", - "# Check deployment status\n", - "deployment_status = client.inference.deployments.retrieve(\n", - " name=deployment.name,\n", - " workspace=\"default\"\n", - ")\n", - "\n", - "print(f\"Deployment name: {deployment.name}\")\n", - "print(f\"Deployment status: {deployment_status.status}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Monitor status of deployment" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import time\n", - "from IPython.display import clear_output\n", - "\n", - "# Poll deployment status every 15 seconds until ready\n", - "TIMEOUT_MINUTES = 30\n", - "start_time = time.time()\n", - "timeout_seconds = TIMEOUT_MINUTES * 60\n", - "\n", - "print(f\"Monitoring deployment '{deployment.name}'...\")\n", - "print(f\"Timeout: {TIMEOUT_MINUTES} minutes\\n\")\n", - "\n", - "while True:\n", - " deployment_status = client.inference.deployments.retrieve(\n", - " name=deployment.name,\n", - " workspace=\"default\"\n", - " )\n", - " \n", - " elapsed = time.time() - start_time\n", - " elapsed_min = int(elapsed // 60)\n", - " elapsed_sec = int(elapsed % 60)\n", - " \n", - " clear_output(wait=True)\n", - " print(f\"Deployment: {deployment.name}\")\n", - " print(f\"Status: {deployment_status.status}\")\n", - " print(f\"Elapsed time: {elapsed_min}m {elapsed_sec}s\")\n", - " \n", - " # Check if deployment is ready\n", - " if deployment_status.status == \"READY\":\n", - " print(\"\\nDeployment is ready!\")\n", - " if not client.models.wait_for_gateway(deployment.name, workspace=\"default\", timeout=60):\n", - " raise RuntimeError(\"Inference gateway did not become ready\")\n", - " break\n", - " \n", - " # Check for failure states\n", - " if deployment_status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n", - " raise RuntimeError(f\"Deployment failed with status: {deployment_status.status}\")\n", - " \n", - " # Check timeout\n", - " if elapsed > timeout_seconds:\n", - " raise TimeoutError(f\"Deployment timeout after {TIMEOUT_MINUTES} minutes\")\n", - " \n", - " time.sleep(15)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "The deployment service automatically:\n", - "- Downloads model weights from the Files service\n", - "- Provisions storage (PVC) for the weights\n", - "- Configures and starts the NIM container\n", - "\n", - "**Multi-GPU Deployment:**\n", - "\n", - "For larger models requiring multiple GPUs, configure parallelism with environment variables:\n", - "\n", - "```python\n", - "deployment_config = client.inference.deployment_configs.create(\n", - " workspace=\"default\",\n", - " name=\"sft-model-config-multigpu\",\n", - " \n", - " nim_deployment={\n", - " \"image_name\": \"nvcr.io/nim/nvidia/llm-nim\",\n", - " \"image_tag\": \"1.13.1\",\n", - " \"gpu\": 2, # Total GPUs\n", - " \"additional_envs\": {\n", - " \"NIM_TENSOR_PARALLEL_SIZE\": \"2\", # Tensor parallelism\n", - " \"NIM_PIPELINE_PARALLEL_SIZE\": \"1\" # Pipeline parallelism\n", - " }\n", - " }\n", - ")\n", - "```" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "**Single-Node Constraint:** Model deployments are limited to a single node. The maximum `gpu` value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.\n", - "\n", - "---\n", - "\n", - "#### GPU Parallelism\n", - "\n", - "By default, NIM uses all GPUs for tensor parallelism (TP). You can customize this behavior using the `NIM_TENSOR_PARALLEL_SIZE` and `NIM_PIPELINE_PARALLEL_SIZE` environment variables.\n", - "\n", - "| Strategy | Description | Best For |\n", - "|----------|-------------|----------|\n", - "| **Tensor Parallel (TP)** | Splits model layers across GPUs | Lowest latency |\n", - "| **Pipeline Parallel (PP)** | Splits model depth across GPUs | Highest throughput |\n", - "\n", - "**Formula:** `gpu` = `NIM_TENSOR_PARALLEL_SIZE` × `NIM_PIPELINE_PARALLEL_SIZE`\n", - "\n", - "---\n", - "\n", - "#### Example Configurations\n", - "\n", - "**Default (TP=8, PP=1) — Lowest Latency**\n", - "```\n", - "\"gpu\": 8\n", - "# NIM automatically sets NIM_TENSOR_PARALLEL_SIZE=8\n", - "```\n", - "\n", - "**Balanced (TP=4, PP=2)**\n", - "```\n", - "\"gpu\": 8,\n", - "\"additional_envs\": {\n", - " \"NIM_TENSOR_PARALLEL_SIZE\": \"4\",\n", - " \"NIM_PIPELINE_PARALLEL_SIZE\": \"2\"\n", - "}\n", - "```\n", - "\n", - "**Throughput Optimized (TP=2, PP=4)**\n", - "```\n", - "\"gpu\": 8,\n", - "\"additional_envs\": {\n", - " \"NIM_TENSOR_PARALLEL_SIZE\": \"2\",\n", - " \"NIM_PIPELINE_PARALLEL_SIZE\": \"4\"\n", - "}\n", - "```" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 9. Evaluate Your Model\n", - "\n", - "After training, evaluate whether your model meets your requirements:\n", - "\n", - "#### Quick Manual Evaluation" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Wait for deployment to be ready, then test\n", - "messages = [\n", - " {\"role\": \"system\", \"content\": \"You are a helpful assistant.\"},\n", - " {\"role\": \"user\", \"content\": \"Write a short email to my colleague.\"}\n", - "]\n", - "\n", - "response = client.inference.gateway.provider.post(\n", - " \"v1/chat/completions\",\n", - " name=deployment.name,\n", - " workspace=\"default\",\n", - " body={\n", - " \"model\": f\"default/{job.spec.output.name}\", # Match the model_name from deployment config\n", - " \"messages\": messages,\n", - " \"temperature\": 0.7,\n", - " \"max_tokens\": 256\n", - " }\n", - ")\n", - "\n", - "# Display prompt and completion\n", - "print(\"=\" * 60)\n", - "print(\"PROMPT\")\n", - "print(\"=\" * 60)\n", - "for msg in messages:\n", - " print(f\"[{msg['role'].upper()}]\")\n", - " print(msg[\"content\"])\n", - " print()\n", - "\n", - "print(\"=\" * 60)\n", - "print(\"COMPLETION\")\n", - "print(\"=\" * 60)\n", - "print(\"[ASSISTANT]\")\n", - "completion = response[\"choices\"][0][\"message\"][\"content\"]\n", - "print(completion)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### Evaluation Best Practices\n", - "\n", - "**Manual Evaluation** (Recommended)\n", - "- Test with real-world examples from your use case\n", - "- Compare responses to base model and expected outputs\n", - "- Verify the model exhibits desired behavior changes\n", - "- Check edge cases and error handling\n", - "\n", - "**What to look for:**\n", - "- ✅ Model follows your desired output format\n", - "- ✅ Applies domain knowledge correctly\n", - "- ✅ Maintains general language capabilities\n", - "- ✅ Avoids unwanted behaviors or biases\n", - "- ❌ Doesn't hallucinate facts not in training data\n", - "- ❌ Doesn't produce repetitive or nonsensical outputs\n", - "\n", - "---\n", - "\n", - "## Hyperparameters\n", - "\n", - "For detailed information on all available hyperparameters, recommended values, and tuning guidance, see the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n", - "\n", - "---\n", - "\n", - "\n", - "## Troubleshooting\n", - "\n", - "**Job fails during model download:**\n", - "- Verify authentication secrets are configured (see [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n", - "- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n", - "- Check the `model_uri` format is correct (`fileset://`)\n", - "- Ensure you have accepted the model's terms of service on HuggingFace\n", - "- Check job status and logs: `client.customization.jobs.retrieve(name=job.name, workspace=\"default\")`\n", - "\n", - "**Job fails with OOM (Out of Memory) error:**\n", - "1. **First try:** Reduce `micro_batch_size` from 2 to 1\n", - "2. **Still OOM:** Reduce `batch_size` from 16 to 8\n", - "3. **Still OOM:** Reduce `max_seq_length` from 2048 to 1024 or 512\n", - "4. **Last resort:** Increase GPU count and use `tensor_parallel_size` for model sharding\n", - "\n", - "**Loss curves not decreasing (underfitting):**\n", - "- Increase training duration: `epochs: 5-10` instead of 3\n", - "- Adjust learning rate: Try `1e-5` to `1e-4`\n", - "- Add warmup: Set `warmup_steps` to ~10% of total training steps\n", - "- Check data quality: Verify formatting, remove duplicates, ensure diversity\n", - "\n", - "**Training loss decreases but validation loss increases (overfitting):**\n", - "- Reduce epochs: Try `epochs: 1-2` instead of 5+\n", - "- Lower learning rate: Use `2e-5` or `1e-5`\n", - "- Increase dataset size and diversity\n", - "- Verify train/validation split has no data leakage\n", - "\n", - "**Model output quality is poor despite good training metrics:**\n", - "- Training metrics optimize for loss, not your actual task—evaluate on real use cases\n", - "- Review data quality, format, and diversity—metrics can be misleading with poor data\n", - "- Try a different base model size or architecture\n", - "- Adjust learning rate and batch size\n", - "- Compare to baseline: Test base model to ensure fine-tuning improved performance\n", - "\n", - "**Deployment fails:**\n", - "- Verify output model exists: `client.models.retrieve(name=job.spec.output.name, workspace=\"default\")`\n", - "- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n", - "- Ensure sufficient GPU resources available for model size\n", - "- Verify NIM image tag `1.13.1` is compatible with your model\n", - "\n", - "\n", - "## Next Steps\n", - "\n", - "- [Monitor training metrics](fine-tune-metrics) in detail\n", - "- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n", - "- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning" - ] - } - ], - "metadata": { - "kernelspec": { - "display_name": ".venv", - "language": "python", - "name": "python3" - }, - "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 3 - }, - "file_extension": ".py", - "mimetype": "text/x-python", - "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython3", - "version": "3.11.14" - } - }, - "nbformat": 4, - "nbformat_minor": 4 -} diff --git a/docs/customizer/tutorials/embedding-customization-job.ipynb b/docs/customizer/tutorials/embedding-customization-job.ipynb index bb61f30218..7071636c51 100644 --- a/docs/customizer/tutorials/embedding-customization-job.ipynb +++ b/docs/customizer/tutorials/embedding-customization-job.ipynb @@ -1,946 +1,989 @@ { - "cells": [ - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "\n", - "\n", - "\n", - "# Embedding Model Customization\n", - "\n", - "Learn how to fine-tune an embedding model to improve retrieval accuracy for your specific domain.\n", - "\n", - "## About\n", - "\n", - "Embedding models convert text into dense vector representations that capture semantic meaning. Fine-tuning these models on your domain data significantly improves retrieval accuracy—in RAG pipelines, this means the LLM receives more relevant context and produces better answers.\n", - "\n", - "**What you will achieve with embedding fine-tuning:**\n", - "\n", - "- 🎯 **Domain specialization:** Adapt general embeddings for legal, medical, scientific, or financial content\n", - "- 📈 **Improved retrieval:** Achieve 6-10% better recall on domain-specific benchmarks\n", - "- 🔍 **Semantic understanding:** Teach the model your domain's vocabulary and relationships\n", - "\n", - "**Recall@5** measures the fraction of relevant documents that appear in the top 5 search results.\n", - "\n", - "**About the baseline:** In retrieval benchmarks like SciDocs, the pretrained model achieves ~0.159 Recall@5. After fine-tuning on scientific paper triplets, you can expect 6-10% improvement (~0.17 Recall@5).\n", - "\n", - "### Dataset Format for Embedding Models\n", - "\n", - "Embedding models require **triplet format** for contrastive learning:\n", - "\n", - "```json\n", - "{\"query\": \"What is machine learning?\", \"pos_doc\": \"Machine learning is a subset of AI...\", \"neg_doc\": [\"Gardening tips for beginners...\"]}\n", - "```\n", - "\n", - "- **`query`**: The search query or question\n", - "- **`pos_doc`**: A document relevant to the query (positive example)\n", - "- **`neg_doc`**: List of hard negatives—documents that share some overlap with the query but are not actually relevant (negative example)\n", - "\n", - "The model learns to maximize similarity between query and positive document while minimizing similarity with negative documents." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Prerequisites\n", - "\n", - "Before starting this tutorial, ensure you have:\n", - "\n", - "1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n", - "2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n", - "3. **HuggingFace token** with read access to download the SPECTER dataset (get one at [huggingface.co/settings/tokens](https://huggingface.co/settings/tokens))\n", - "4. **NGC API key** to pull NIM container images from nvcr.io (get one at [ngc.nvidia.com](https://ngc.nvidia.com/) → Setup → Generate API Key)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Quick Start\n", - "\n", - "### 1. Initialize SDK\n", - "\n", - "The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](../../get-started/quickstart.md) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable:\n", - "\n", - "```sh\n", - "export NMP_BASE_URL=\n", - "```" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import json\n", - "import os\n", - "from nemo_platform import NeMoPlatform, ConflictError\n", - "\n", - "NMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\n", - "client = NeMoPlatform(\n", - " base_url=NMP_BASE_URL,\n", - " workspace=\"default\"\n", - ")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 2. Establish Baseline Performance\n", - "\n", - "Before fine-tuning, establish baseline performance with the pretrained model. Deploy it, run a test query, and observe where it struggles. Following fine-tuning, compare the results.\n", - "\n", - "**Scenario:** Searching scientific papers by meaning, not keywords.\n", - "\n", - "**Demo setup:**\n", - "- **Query:** \"Conditional Random Fields\" (CRFs) - a method for sequence labeling in NLP\n", - "- **Trap:** \"Random Forests\" shares the word \"random\" but is an unrelated tree-based algorithm\n", - "- **Goal:** The model should distinguish between them." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Install required packages for dataset preparation\n", - "%pip install -q datasets huggingface_hub" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import uuid\n", - "import numpy as np\n", - "\n", - "# Demo query and documents for baseline comparison\n", - "DEMO_QUERY = \"Conditional Random Fields: Probabilistic Models for Segmenting and Labeling Sequence Data\"\n", - "\n", - "DEMO_DOCS = [\n", - " \"Bidirectional LSTM-CRF Models for Sequence Tagging\", # CRF-based paper\n", - " \"An Introduction to Conditional Random Fields\", # CRF tutorial \n", - " \"Random Forests\", # Keyword trap! Unrelated.\n", - " \"Neural Architectures for Named Entity Recognition\", # Related to sequence labeling; may use CRFs\n", - " \"Support Vector Machines for Classification\", # Unrelated ML method\n", - "]\n", - "\n", - "DEMO_LABELS = [\"BiLSTM-CRF\", \"CRF Tutorial\", \"Random Forest\", \"NER\", \"SVM\"]\n", - "DEMO_RELEVANT = {0, 1, 3} # Papers actually relevant to CRFs\n", - "\n", - "def cosine_similarity(a, b):\n", - " \"\"\"Calculate cosine similarity between two vectors.\"\"\"\n", - " return np.dot(a, b) / (np.linalg.norm(a) * np.linalg.norm(b))" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# NGC API key is required to pull NIM images from nvcr.io\nNGC_API_KEY = os.environ.get(\"NGC_API_KEY\")\nif not NGC_API_KEY:\n raise ValueError(\"NGC_API_KEY environment variable is required. Get one at https://ngc.nvidia.com/ → Setup → Generate API Key\")\n\n# Create NGC secret for pulling NIM images\nNGC_SECRET_NAME = \"ngc-api-key\"\ntry:\n client.secrets.create(name=NGC_SECRET_NAME, workspace=\"default\", value=NGC_API_KEY)\n print(f\"Created secret: {NGC_SECRET_NAME}\")\nexcept ConflictError:\n print(f\"Secret '{NGC_SECRET_NAME}' already exists, continuing...\")\n\n# Deploy base model for baseline comparison\nBASE_MODEL_HF = \"nvidia/llama-nemotron-embed-1b-v2\"\nNIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\nNIM_TAG = \"1.13.0\"\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f\"baseline-embedding-cfg-{baseline_suffix}\"\nBASELINE_DEPLOYMENT_NAME = f\"baseline-embedding-{baseline_suffix}\"\n\nprint(\"Creating baseline deployment config...\")\nbaseline_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name=NIM_IMAGE,\n image_tag=NIM_TAG,\n gpu=1,\n image_pull_secret=NGC_SECRET_NAME,\n )\n)\n\nprint(\"Deploying base model...\")\nbaseline_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_config.name\n)\nprint(f\"Baseline deployment: {baseline_deployment.name}\")" - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import time\n", - "from IPython.display import clear_output\n", - "\n", - "# Wait for baseline deployment\n", - "TIMEOUT_MINUTES = 15\n", - "start_time = time.time()\n", - "\n", - "print(f\"Waiting for baseline deployment...\")\n", - "while True:\n", - " status = client.inference.deployments.retrieve(\n", - " name=BASELINE_DEPLOYMENT_NAME,\n", - " workspace=\"default\"\n", - " )\n", - " \n", - " elapsed = time.time() - start_time\n", - " elapsed_str = f\"{int(elapsed//60)}m {int(elapsed%60)}s\"\n", - " \n", - " clear_output(wait=True)\n", - " print(f\"Baseline deployment: {status.status} | {elapsed_str}\")\n", - " \n", - " if status.status == \"READY\":\n", - " print(\"Baseline model ready!\")\n", - " if not client.models.wait_for_gateway(BASELINE_DEPLOYMENT_NAME, workspace=\"default\", timeout=60):\n", - " raise RuntimeError(\"Inference gateway did not become ready\")\n", - " break\n", - " if status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n", - " raise RuntimeError(f\"Baseline deployment failed: {status.status}\")\n", - " if elapsed > TIMEOUT_MINUTES * 60:\n", - " raise TimeoutError(\"Baseline deployment timeout\")\n", - " \n", - " time.sleep(10)\n", - "\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Run baseline ranking with the base model\n", - "BASE_MODEL_ID = \"nvidia/llama-nemotron-embed-1b-v2\"\n", - "\n", - "# Get query embedding\n", - "query_response = client.inference.gateway.provider.post(\n", - " \"v1/embeddings\",\n", - " name=BASELINE_DEPLOYMENT_NAME,\n", - " workspace=\"default\",\n", - " body={\n", - " \"model\": BASE_MODEL_ID,\n", - " \"input\": [DEMO_QUERY],\n", - " \"input_type\": \"query\"\n", - " }\n", - ")\n", - "base_query_emb = query_response[\"data\"][0][\"embedding\"]\n", - "\n", - "# Get document embeddings\n", - "doc_response = client.inference.gateway.provider.post(\n", - " \"v1/embeddings\",\n", - " name=BASELINE_DEPLOYMENT_NAME,\n", - " workspace=\"default\",\n", - " body={\n", - " \"model\": BASE_MODEL_ID,\n", - " \"input\": DEMO_DOCS,\n", - " \"input_type\": \"passage\"\n", - " }\n", - ")\n", - "base_doc_embs = [d[\"embedding\"] for d in doc_response[\"data\"]]\n", - "\n", - "# Calculate similarities and rank\n", - "scores = [(i, cosine_similarity(base_query_emb, base_doc_embs[i])) for i in range(len(DEMO_DOCS))]\n", - "BASELINE_RANKING = sorted(scores, key=lambda x: -x[1])\n", - "\n", - "# Display baseline results\n", - "print(f\"Query: \\\"{DEMO_QUERY}\\\"\\n\")\n", - "print(\"Base Model Ranking:\")\n", - "print(\"-\" * 55)\n", - "for rank, (idx, score) in enumerate(BASELINE_RANKING, 1):\n", - " marker = \" <-- relevant\" if idx in DEMO_RELEVANT else \"\"\n", - " print(f\" #{rank} [{score:.3f}] {DEMO_LABELS[idx]}{marker}\")" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Delete baseline deployment to free GPU for training\n", - "print(\"Deleting baseline deployment to free GPU...\")\n", - "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace=\"default\")\n", - "\n", - "# Wait for deployment to be fully deleted before deleting config\n", - "print(\"Waiting for deployment deletion...\")\n", - "while True:\n", - " try:\n", - " status = client.inference.deployments.retrieve(name=BASELINE_DEPLOYMENT_NAME, workspace=\"default\")\n", - " if status.status == \"DELETED\":\n", - " break\n", - " print(f\" Status: {status.status}\")\n", - " time.sleep(5)\n", - " except Exception:\n", - " # Deployment no longer exists\n", - " break\n", - "\n", - "# Now safe to delete the config\n", - "client.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace=\"default\")\n", - "print(\"GPU freed. Proceed to fine-tune and improve these rankings.\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 3. Prepare Dataset\n", - "\n", - "Use the [SPECTER dataset](https://huggingface.co/datasets/embedding-data/SPECTER) from HuggingFace, a collection of scientific paper triplets where papers that cite each other are considered related.\n", - "\n", - "**Dataset structure:**\n", - "- ~684K scientific paper triplets (this tutorial uses 10%)\n", - "- Each triplet: (query paper, positive/related paper, negative/unrelated paper)\n", - "- Papers that cite each other are marked as \"related\"\n", - "\n", - "In this tutorial the following dataset directory structure will be used:\n", - "```\n", - "embedding-dataset\n", - "`-- training.jsonl\n", - "`-- validation.jsonl\n", - "```" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 4. Download and Format SPECTER Dataset\n", - "\n", - "The SPECTER dataset requires conversion to the triplet format required for fine-tuning." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from pathlib import Path\n", - "from datasets import load_dataset\n", - "import json\n", - "\n", - "# HuggingFace token for dataset access\n", - "HF_TOKEN = os.environ.get(\"HF_TOKEN\")\n", - "if not HF_TOKEN:\n", - " raise ValueError(\"HF_TOKEN environment variable is required. Get one at https://huggingface.co/settings/tokens\")\n", - "os.environ[\"HF_TOKEN\"] = HF_TOKEN\n", - "\n", - "# Configuration\n", - "DATASET_SIZE = 3000 # Number of triplets (increase for better results, max ~684K)\n", - "VALIDATION_SPLIT = 0.05 # 5% held out for validation\n", - "SEED = 42\n", - "DATASET_PATH = Path(\"embedding-dataset\").absolute()\n", - "\n", - "# Create directory\n", - "os.makedirs(DATASET_PATH, exist_ok=True)\n", - "\n", - "# Download SPECTER dataset\n", - "print(\"Downloading SPECTER dataset...\")\n", - "data = load_dataset(\"embedding-data/SPECTER\")[\"train\"].shuffle(seed=SEED).select(range(DATASET_SIZE))\n", - "\n", - "# Split into train/validation\n", - "print(\"Splitting into train/validation...\")\n", - "splits = data.train_test_split(test_size=VALIDATION_SPLIT, seed=SEED)\n", - "train_data = splits[\"train\"]\n", - "validation_data = splits[\"test\"]\n", - "\n", - "# Convert to triplet JSONL format\n", - "print(\"Saving to JSONL...\")\n", - "for name, dataset in [(\"training\", train_data), (\"validation\", validation_data)]:\n", - " with open(f\"{DATASET_PATH}/{name}.jsonl\", \"w\") as f:\n", - " for row in dataset:\n", - " # SPECTER format: row['set'] = [query, positive, negative]\n", - " triplet = {\n", - " \"query\": row[\"set\"][0],\n", - " \"pos_doc\": row[\"set\"][1],\n", - " \"neg_doc\": [row[\"set\"][2]] # List of negative documents\n", - " }\n", - " f.write(json.dumps(triplet) + \"\\n\")\n", - "\n", - "print(f\"\\nPrepared {len(train_data):,} training, {len(validation_data):,} validation samples\")\n", - "print(f\"\\nExample triplet:\")\n", - "print(f\" Query: {train_data[0]['set'][0][:100]}...\")\n", - "print(f\" Positive: {train_data[0]['set'][1][:100]}...\")\n", - "print(f\" Negative: {train_data[0]['set'][2][:100]}...\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 5. Create Dataset FileSet and Upload Training Data" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Create fileset to store embedding training data\n", - "DATASET_NAME = \"embedding-dataset\"\n", - "\n", - "try:\n", - " client.files.filesets.create(\n", - " workspace=\"default\",\n", - " name=DATASET_NAME,\n", - " description=\"SPECTER embedding training data (scientific paper triplets)\"\n", - " )\n", - " print(f\"Created fileset: {DATASET_NAME}\")\n", - "except ConflictError:\n", - " print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n", - "\n", - "# Upload training data files\n", - "client.files.upload(\n", - " local_path=DATASET_PATH,\n", - " remote_path=\"\",\n", - " fileset=DATASET_NAME,\n", - " workspace=\"default\"\n", - ")\n", - "\n", - "# Validate upload\n", - "print(\"\\nUploaded files:\")\n", - "print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 6. Secrets Setup\n", - "\n", - "Configure authentication for accessing base models:\n", - "\n", - "- **NGC models** (`ngc://` URIs): Requires NGC API key\n", - "- **HuggingFace models** (`hf://` URIs): Requires HF token for gated/private models\n", - "\n", - "Get your credentials:\n", - "- [NGC API Key](https://ngc.nvidia.com/) (Setup → Generate API Key)\n", - "- [HuggingFace Token](https://huggingface.co/settings/tokens) (Create token with Read access)\n", - "\n", - "---\n", - "\n", - "#### Quick Setup Example\n", - "\n", - "This tutorial fine-tunes [nvidia/llama-3.2-nv-embedqa-1b-v2](https://huggingface.co/nvidia/llama-3.2-nv-embedqa-1b-v2), NVIDIA embedding model optimized for question-answering and retrieval tasks." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": "# Create secrets for model access\n# Note: NGC_API_KEY secret was already created in the baseline step (Step 2)\nHF_TOKEN = os.getenv(\"HF_TOKEN\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n name=name,\n workspace=\"default\",\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name, workspace=\"default\")\n\n\n# Create HuggingFace token secret (for downloading model from HF during training)\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(f\"HF_TOKEN secret: {hf_secret.name}\")\n\n# NGC secret was already created in baseline step\nprint(f\"NGC_API_KEY secret: {NGC_SECRET_NAME} (created in Step 2)\")" - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 7. Create Base Model FileSet and Model Entity\n", - "\n", - "Create a fileset pointing to the [nvidia/llama-3.2-nv-embedqa-1b-v2](https://huggingface.co/nvidia/llama-3.2-nv-embedqa-1b-v2) embedding model from HuggingFace, then create a Model Entity that references this fileset. Model downloading will take place at training time.\n", - "\n", - "---\n", - "*Note*: Either `MODEL_NAME` or `HF_REPO_ID` below must contain the substring embed to indicate that this is an embedding model." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import time\n", - "from nemo_platform.types.files import HuggingfaceStorageConfigParam\n", - "\n", - "HF_REPO_ID = \"nvidia/llama-nemotron-embed-1b-v2\"\n", - "MODEL_NAME = \"nv-nemotron-embed-1b-base\"\n", - "\n", - "# Ensure you have a HuggingFace token secret created\n", - "try:\n", - " base_model_fs = client.files.filesets.create(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " description=\"NVIDIA Llama 3.2 NV EmbedQA 1B v2 embedding model\",\n", - " storage=HuggingfaceStorageConfigParam(\n", - " type=\"huggingface\",\n", - " # repo_id is the full model name from Hugging Face\n", - " repo_id=HF_REPO_ID,\n", - " repo_type=\"model\",\n", - " # we use the secret created in the previous step\n", - " token_secret=hf_secret.name\n", - " )\n", - " )\n", - "except ConflictError as e:\n", - " print(f\"Base model fileset already exists. Skipping creation.\")\n", - " base_model_fs = client.files.filesets.retrieve(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " )\n", - "\n", - "# Create Model Entity referencing the FileSet\n", - "try:\n", - " base_model = client.models.create(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " fileset=f\"default/{MODEL_NAME}\",\n", - " trust_remote_code=True,\n", - " )\n", - " print(f\"Created Model Entity: {MODEL_NAME}\")\n", - "except ConflictError:\n", - " print(f\"Base model already exists. Updating fileset if different.\")\n", - " base_model = client.models.update(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " fileset=f\"default/{MODEL_NAME}\",\n", - " trust_remote_code=True,\n", - " )\n", - "\n", - "print(f\"\\nBase model fileset: fileset://default/{base_model.name}\")\n", - "print(\"\\nBase model files:\")\n", - "print(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=\"default\").data], indent=2))\n", - "\n", - "# Wait for ModelSpec to be populated from the checkpoint\n", - "print(\"\\nWaiting for ModelSpec to be populated...\")\n", - "SPEC_TIMEOUT_SECONDS = 120\n", - "spec_start = time.time()\n", - "while not base_model.spec:\n", - " if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n", - " raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n", - " time.sleep(2)\n", - " base_model = client.models.retrieve(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " )\n", - "\n", - "print(f\"ModelSpec populated: {base_model.spec}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 8. Create Embedding Fine-tuning Job\n", - "\n", - "Create a customization job to fine-tune the embedding model using contrastive learning on the SPECTER dataset.\n", - "\n", - "**Key hyperparameters for embedding fine-tuning:**\n", - "- **`training_type`**: `sft` (supervised fine-tuning)\n", - "- **Full fine-tuning**: No `peft` config needed (omit for all-weights training)\n", - "- **`learning_rate`**: Lower values (1e-6 to 5e-6) work well for embedding models\n", - "- **`batch_size`**: Larger batches improve contrastive learning (128-256 recommended)\n", - "\n", - "**NOTE:**\n", - "\n", - "NeMo Platform does not support unmerged LoRA adapters for embedding models because the embedding NIM requires ONNX format, which cannot represent standalone adapters. This notebook creates a job with all-weights finetuning but you can also run LoRA with `merge=True`, which trains a LoRA adapter and then merges it back into the base model after training. The final output is a standard full-weight checkpoint, identical in format to an all-weights fine-tuned model, but LoRA training is faster, uses less memory, and is more lenient in hyperparameter tuning.\n", - "\n", - "To do that update the job request like so\n", - "\n", - "```\n", - "job = client.customization.jobs.create(\n", - " name=JOB_NAME,\n", - " workspace=\"default\",\n", - " spec=CustomizationJobInputParam(\n", - " model=f\"default/{base_model.name}\",\n", - " dataset=f\"fileset://default/{DATASET_NAME}\",\n", - " training=SftTrainingParam(\n", - " type=\"sft\",\n", - " epochs=EPOCHS,\n", - " batch_size=BATCH_SIZE,\n", - " learning_rate=LEARNING_RATE,\n", - " max_seq_length=MAX_SEQ_LENGTH,\n", - " micro_batch_size=1,\n", - " parallelism=ParallelismParamsParam(\n", - " num_gpus_per_node=1,\n", - " num_nodes=1,\n", - " tensor_parallel_size=1,\n", - " pipeline_parallel_size=1,\n", - " ),\n", - " peft=LoRaParamsParam(\n", - " type= \"lora\",\n", - " merge=True\n", - " )\n", - " )\n", - " )\n", - ")\n", - "```" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from nemo_platform.types.customization import (\n", - " CustomizationJobInputParam,\n", - " SftTrainingParam,\n", - " ParallelismParamsParam\n", - ")\n", - "\n", - "job_suffix = uuid.uuid4().hex[:4]\n", - "JOB_NAME = f\"embedding-finetune-job-{job_suffix}\"\n", - "\n", - "# Hyperparameters optimized for embedding fine-tuning\n", - "EPOCHS = 1\n", - "BATCH_SIZE = 128 # Larger batches help contrastive learning\n", - "LEARNING_RATE = 5e-6 # Lower LR for embedding models\n", - "MAX_SEQ_LENGTH = 512 # Typical for embedding models\n", - "\n", - "# Note: The 'name' field must contain 'embed' for the customizer to detect this as an embedding model\n", - "job = client.customization.jobs.create(\n", - " name=JOB_NAME,\n", - " workspace=\"default\",\n", - " spec=CustomizationJobInputParam(\n", - " model=f\"default/{base_model.name}\",\n", - " dataset=f\"fileset://default/{DATASET_NAME}\",\n", - " training=SftTrainingParam(\n", - " type=\"sft\",\n", - " epochs=EPOCHS,\n", - " batch_size=BATCH_SIZE,\n", - " learning_rate=LEARNING_RATE,\n", - " max_seq_length=MAX_SEQ_LENGTH,\n", - " micro_batch_size=1,\n", - " parallelism=ParallelismParamsParam(\n", - " num_gpus_per_node=1,\n", - " num_nodes=1,\n", - " tensor_parallel_size=1,\n", - " pipeline_parallel_size=1,\n", - " ),\n", - " )\n", - " )\n", - ")\n", - "\n", - "print(f\"Job ID: {job.name}\")\n", - "print(f\"Output model: {job.spec.output.name}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 9. Track Training Progress" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import time\n", - "from IPython.display import clear_output\n", - "\n", - "# Poll job status every 10 seconds until completed\n", - "while True:\n", - " status = client.customization.jobs.get_status(\n", - " name=job.name,\n", - " workspace=\"default\"\n", - " )\n", - " \n", - " clear_output(wait=True)\n", - " print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n", - "\n", - " # Extract training progress from nested steps structure\n", - " step: int | None = None\n", - " max_steps: int | None = None\n", - " training_phase: str | None = None\n", - "\n", - " for job_step in status.steps or []:\n", - " if job_step.name == \"customization-training-job\":\n", - " for task in job_step.tasks or []:\n", - " task_details = task.status_details or {}\n", - " step = task_details.get(\"step\")\n", - " max_steps = task_details.get(\"max_steps\")\n", - " training_phase = task_details.get(\"phase\")\n", - " break\n", - " break\n", - "\n", - " if step is not None and max_steps is not None:\n", - " progress_pct = (step / max_steps) * 100\n", - " print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n", - " if training_phase:\n", - " print(f\"Training Phase: {training_phase}\")\n", - " else:\n", - " print(\"Training step not started yet or progress info not available\")\n", - " \n", - " # Exit loop when job is completed (or failed/cancelled)\n", - " if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n", - " print(f\"\\nJob finished with status: {status.status}\")\n", - " break\n", - " \n", - " time.sleep(10)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "**Interpreting Embedding Training Metrics:**\n", - "\n", - "Embedding models use contrastive loss—lower values indicate better separation between similar and dissimilar pairs:\n", - "\n", - "| Scenario | Interpretation | Action |\n", - "|----------|----------------|--------|\n", - "| **Loss steadily decreasing** | Model learning semantic relationships | Continue training |\n", - "| **Loss plateaus early** | May need more data or epochs | Increase dataset/epochs |\n", - "| **Loss spikes** | Training instability | Lower learning rate |\n", - "| **Validation loss increasing** | Overfitting | Reduce epochs, add data |" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 10. Deploy Fine-Tuned Embedding Model\n", - "\n", - "After training completes, deploy the embedding model using the Deployment Management Service:" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Validate model entity exists\n", - "model_entity = client.models.retrieve(workspace=\"default\", name=job.spec.output.name)\n", - "print(model_entity.model_dump_json(indent=2))" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from nemo_platform.types.inference import NIMDeploymentParam\n", - "\n", - "# Create deployment config for embedding model\n", - "deploy_suffix = uuid.uuid4().hex[:4]\n", - "DEPLOYMENT_CONFIG_NAME = f\"embedding-model-deployment-cfg-{deploy_suffix}\"\n", - "DEPLOYMENT_NAME = f\"embedding-model-deployment-{deploy_suffix}\"\n", - "\n", - "# Embedding NIM image\n", - "NIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\n", - "NIM_TAG = \"1.13.0\" # Update if using newer NIM release\n", - "\n", - "deployment_config = client.inference.deployment_configs.create(\n", - " workspace=\"default\",\n", - " name=DEPLOYMENT_CONFIG_NAME,\n", - " nim_deployment=NIMDeploymentParam(\n", - " image_name=NIM_IMAGE,\n", - " image_tag=NIM_TAG,\n", - " gpu=1,\n", - " model_name=job.spec.output.name,\n", - " model_namespace=\"default\",\n", - " )\n", - ")\n", - "\n", - "# Deploy model\n", - "deployment = client.inference.deployments.create(\n", - " workspace=\"default\",\n", - " name=DEPLOYMENT_NAME,\n", - " config=deployment_config.name\n", - ")\n", - "\n", - "print(f\"Deployment name: {deployment.name}\")\n", - "print(f\"Deployment status: {client.inference.deployments.retrieve(name=deployment.name, workspace='default').status}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### Track Deployment Status" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import time\n", - "from IPython.display import clear_output\n", - "\n", - "# Poll deployment status every 15 seconds until ready\n", - "TIMEOUT_MINUTES = 30\n", - "start_time = time.time()\n", - "timeout_seconds = TIMEOUT_MINUTES * 60\n", - "\n", - "print(f\"Monitoring deployment '{deployment.name}'...\")\n", - "print(f\"Timeout: {TIMEOUT_MINUTES} minutes\\n\")\n", - "\n", - "while True:\n", - " deployment_status = client.inference.deployments.retrieve(\n", - " name=deployment.name,\n", - " workspace=\"default\"\n", - " )\n", - " \n", - " elapsed = time.time() - start_time\n", - " elapsed_min = int(elapsed // 60)\n", - " elapsed_sec = int(elapsed % 60)\n", - " \n", - " clear_output(wait=True)\n", - " print(f\"Deployment: {deployment.name}\")\n", - " print(f\"Status: {deployment_status.status}\")\n", - " print(f\"Elapsed time: {elapsed_min}m {elapsed_sec}s\")\n", - " \n", - " # Check if deployment is ready\n", - " if deployment_status.status == \"READY\":\n", - " print(\"\\nDeployment is ready!\")\n", - " if not client.models.wait_for_gateway(deployment.name, workspace=\"default\", timeout=60):\n", - " raise RuntimeError(\"Inference gateway did not become ready\")\n", - " break\n", - " \n", - " # Check for failure states\n", - " if deployment_status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n", - " raise RuntimeError(f\"Deployment failed with status: {deployment_status.status}\")\n", - " \n", - " # Check timeout\n", - " if elapsed > timeout_seconds:\n", - " raise TimeoutError(f\"Deployment timeout after {TIMEOUT_MINUTES} minutes\")\n", - " \n", - " time.sleep(15)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 11. View the Improvement\n", - "\n", - "Run the same query against the fine-tuned model and compare to the baseline from earlier." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Compare: same query, base model vs fine-tuned\n", - "# Using the same DEMO_QUERY and DEMO_DOCS from the baseline test\n", - "MODEL_ID = f\"default/{job.spec.output.name}\"\n", - "\n", - "# Get query embedding from fine-tuned model\n", - "query_response = client.inference.gateway.provider.post(\n", - " \"v1/embeddings\",\n", - " name=deployment.name,\n", - " workspace=\"default\",\n", - " body={\n", - " \"model\": MODEL_ID,\n", - " \"input\": [DEMO_QUERY],\n", - " \"input_type\": \"query\"\n", - " }\n", - ")\n", - "query_embedding = query_response[\"data\"][0][\"embedding\"]\n", - "\n", - "# Get document embeddings from fine-tuned model\n", - "doc_response = client.inference.gateway.provider.post(\n", - " \"v1/embeddings\",\n", - " name=deployment.name,\n", - " workspace=\"default\",\n", - " body={\n", - " \"model\": MODEL_ID,\n", - " \"input\": DEMO_DOCS,\n", - " \"input_type\": \"passage\"\n", - " }\n", - ")\n", - "doc_embeddings = [d[\"embedding\"] for d in doc_response[\"data\"]]\n", - "\n", - "# Calculate similarities and rank\n", - "scores = [(i, cosine_similarity(query_embedding, doc_embeddings[i])) for i in range(len(DEMO_DOCS))]\n", - "FINETUNED_RANKING = sorted(scores, key=lambda x: -x[1])\n", - "\n", - "# Display side-by-side comparison\n", - "print(f\"Query: \\\"{DEMO_QUERY}\\\"\\n\")\n", - "print(f\"{'Rank':<6} {'Base Model':<30} {'Fine-tuned Model':<30}\")\n", - "print(\"-\" * 66)\n", - "\n", - "for rank in range(len(DEMO_DOCS)):\n", - " b_idx, b_score = BASELINE_RANKING[rank]\n", - " f_idx, f_score = FINETUNED_RANKING[rank]\n", - " \n", - " b_label = f\"{DEMO_LABELS[b_idx]} [{b_score:.3f}]\" + (\" *\" if b_idx in DEMO_RELEVANT else \"\")\n", - " f_label = f\"{DEMO_LABELS[f_idx]} [{f_score:.3f}]\" + (\" *\" if f_idx in DEMO_RELEVANT else \"\")\n", - " \n", - " print(f\"#{rank+1:<5} {b_label:<30} {f_label:<30}\")\n", - "\n", - "print(\"\\n* = relevant paper\")\n", - "print(\"\\nThe fine-tuned model pushes 'Random Forest' down and ranks CRF papers higher.\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Evaluation Best Practices\n", - "\n", - "**Manual Evaluation** (Recommended)\n", - "- Test with real-world queries from your domain\n", - "- Compare retrieval rankings before and after fine-tuning\n", - "- Check that semantically similar items rank higher than keyword matches\n", - "\n", - "**What to look for:**\n", - "- ✅ Relevant documents consistently rank in top positions\n", - "- ✅ Keyword traps (like \"Random Forest\" vs \"Random Fields\") are handled correctly\n", - "- ✅ Domain-specific terminology is understood\n", - "- ❌ Unrelated documents with matching keywords do not rank high\n", - "\n", - "**Benchmark Evaluation**\n", - "\n", - "For systematic evaluation, use the NeMo Evaluator service with retrieval benchmarks like SciDocs, BEIR, or MTEB. Refer to the [Evaluator documentation](../../evaluator/index.md) for details.\n", - "\n", - "---\n", - "\n", - "## Hyperparameters\n", - "\n", - "For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n", - "\n", - "**Embedding-Specific Recommendations:**\n", - "\n", - "| Parameter | Recommended | Notes |\n", - "|-----------|-------------|-------|\n", - "| `learning_rate` | 1e-6 to 5e-6 | Lower than standard SFT |\n", - "| `batch_size` | 128-256 | Larger batches improve contrastive learning |\n", - "| `max_seq_length` | 512 | Typical for embedding models |\n", - "| `epochs` | 1-3 | Start small, increase if needed |\n", - "\n", - "---\n", - "\n", - "## Troubleshooting\n", - "\n", - "**Embeddings do not show improved retrieval:**\n", - "- Verify dataset quality: triplets should have clear positive/negative distinctions\n", - "- Use hard negatives: negatives should share some overlap with the query but not be relevant (easy negatives do not teach the model much)\n", - "- Increase dataset size: 10K+ triplets recommended for meaningful improvement\n", - "- Try more epochs: embedding models often need multiple passes\n", - "- Lower learning rate: embedding models are sensitive to LR\n", - "\n", - "**Training loss not decreasing:**\n", - "- Check triplet format: ensure `neg_doc` is a list even for single negatives\n", - "- Verify hard negative quality: negatives should be challenging but clearly non-relevant\n", - "- Increase batch size: contrastive learning benefits from larger batches\n", - "\n", - "**Deployment fails:**\n", - "- Ensure you use the correct NIM image for embedding models\n", - "- Verify sufficient GPU memory for the model size\n", - "- Check deployment status: `client.inference.deployments.retrieve(name=deployment.name, workspace=\"default\")` and refer to platform logs for debugging\n", - "\n", - "## Next Steps\n", - "\n", - "- [Monitor training metrics](../manage-customization-jobs/get-job-status.md) in detail\n", - "- [Evaluate your model](../../evaluator/index.md) with retrieval benchmarks\n", - "- Integrate the fine-tuned embedding model into your RAG pipeline\n", - "- Scale up training with the full SPECTER dataset (~684K triplets) for better results" - ] - } - ], - "metadata": { - "kernelspec": { - "display_name": ".venv", - "language": "python", - "name": "python3" - }, - "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 3 - }, - "file_extension": ".py", - "mimetype": "text/x-python", - "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython3", - "version": "3.11.14" - } - }, - "nbformat": 4, - "nbformat_minor": 2 -} + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "\n", + "\n", + "# Embedding Model Customization\n", + "\n", + "Learn how to fine-tune an embedding model to improve retrieval accuracy for your specific domain.\n", + "\n", + "## About\n", + "\n", + "Embedding models convert text into dense vector representations that capture semantic meaning. Fine-tuning these models on your domain data significantly improves retrieval accuracy—in RAG pipelines, this means the LLM receives more relevant context and produces better answers.\n", + "\n", + "**What you will achieve with embedding fine-tuning:**\n", + "\n", + "- 🎯 **Domain specialization:** Adapt general embeddings for legal, medical, scientific, or financial content\n", + "- 📈 **Improved retrieval:** Achieve 6-10% better recall on domain-specific benchmarks\n", + "- 🔍 **Semantic understanding:** Teach the model your domain's vocabulary and relationships\n", + "\n", + "**Recall@5** measures the fraction of relevant documents that appear in the top 5 search results.\n", + "\n", + "**About the baseline:** In retrieval benchmarks like SciDocs, the pretrained model achieves ~0.159 Recall@5. After fine-tuning on scientific paper triplets, you can expect 6-10% improvement (~0.17 Recall@5).\n", + "\n", + "### Dataset Format for Embedding Models\n", + "\n", + "Embedding models require **triplet format** for contrastive learning:\n", + "\n", + "```json\n", + "{\"query\": \"What is machine learning?\", \"pos_doc\": \"Machine learning is a subset of AI...\", \"neg_doc\": [\"Gardening tips for beginners...\"]}\n", + "```\n", + "\n", + "- **`query`**: The search query or question\n", + "- **`pos_doc`**: A document relevant to the query (positive example)\n", + "- **`neg_doc`**: List of hard negatives—documents that share some overlap with the query but are not actually relevant (negative example)\n", + "\n", + "The model learns to maximize similarity between query and positive document while minimizing similarity with negative documents." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Prerequisites\n", + "\n", + "Before starting this tutorial, ensure you have:\n", + "\n", + "1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n", + "2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n", + "3. **HuggingFace token** with read access to download the SPECTER dataset (get one at [huggingface.co/settings/tokens](https://huggingface.co/settings/tokens))\n", + "4. **NGC API key** to pull NIM container images from nvcr.io (get one at [ngc.nvidia.com](https://ngc.nvidia.com/) → Setup → Generate API Key)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Quick Start\n", + "\n", + "### 1. Initialize SDK\n", + "\n", + "The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](../../get-started/quickstart.md) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable:\n", + "\n", + "```sh\n", + "export NMP_BASE_URL=\n", + "```" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "import json\n", + "import os\n", + "from nemo_platform import NeMoPlatform, ConflictError\n", + "\n", + "NMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\n", + "client = NeMoPlatform(\n", + " base_url=NMP_BASE_URL,\n", + " workspace=\"default\"\n", + ")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2. Establish Baseline Performance\n", + "\n", + "Before fine-tuning, establish baseline performance with the pretrained model. Deploy it, run a test query, and observe where it struggles. Following fine-tuning, compare the results.\n", + "\n", + "**Scenario:** Searching scientific papers by meaning, not keywords.\n", + "\n", + "**Demo setup:**\n", + "- **Query:** \"Conditional Random Fields\" (CRFs) - a method for sequence labeling in NLP\n", + "- **Trap:** \"Random Forests\" shares the word \"random\" but is an unrelated tree-based algorithm\n", + "- **Goal:** The model should distinguish between them." + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "# Install required packages for dataset preparation\n", + "%pip install -q datasets huggingface_hub" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "import uuid\n", + "import numpy as np\n", + "\n", + "# Demo query and documents for baseline comparison\n", + "DEMO_QUERY = \"Conditional Random Fields: Probabilistic Models for Segmenting and Labeling Sequence Data\"\n", + "\n", + "DEMO_DOCS = [\n", + " \"Bidirectional LSTM-CRF Models for Sequence Tagging\", # CRF-based paper\n", + " \"An Introduction to Conditional Random Fields\", # CRF tutorial \n", + " \"Random Forests\", # Keyword trap! Unrelated.\n", + " \"Neural Architectures for Named Entity Recognition\", # Related to sequence labeling; may use CRFs\n", + " \"Support Vector Machines for Classification\", # Unrelated ML method\n", + "]\n", + "\n", + "DEMO_LABELS = [\"BiLSTM-CRF\", \"CRF Tutorial\", \"Random Forest\", \"NER\", \"SVM\"]\n", + "DEMO_RELEVANT = {0, 1, 3} # Papers actually relevant to CRFs\n", + "\n", + "def cosine_similarity(a, b):\n", + " \"\"\"Calculate cosine similarity between two vectors.\"\"\"\n", + " return np.dot(a, b) / (np.linalg.norm(a) * np.linalg.norm(b))" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "# NGC API key is required to pull NIM images from nvcr.io\n", + "NGC_API_KEY = os.environ.get(\"NGC_API_KEY\")\n", + "if not NGC_API_KEY:\n", + " raise ValueError(\"NGC_API_KEY environment variable is required. Get one at https://ngc.nvidia.com/ → Setup → Generate API Key\")\n", + "\n", + "# Create NGC secret for pulling NIM images\n", + "NGC_SECRET_NAME = \"ngc-api-key\"\n", + "try:\n", + " client.secrets.create(name=NGC_SECRET_NAME, workspace=\"default\", value=NGC_API_KEY)\n", + " print(f\"Created secret: {NGC_SECRET_NAME}\")\n", + "except ConflictError:\n", + " print(f\"Secret '{NGC_SECRET_NAME}' already exists, continuing...\")\n", + "\n", + "# Deploy base model for baseline comparison\n", + "BASE_MODEL_HF = \"nvidia/llama-nemotron-embed-1b-v2\"\n", + "NIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\n", + "NIM_TAG = \"1.13.0\"\n", + "\n", + "baseline_suffix = uuid.uuid4().hex[:4]\n", + "BASELINE_DEPLOYMENT_CONFIG = f\"baseline-embedding-cfg-{baseline_suffix}\"\n", + "BASELINE_DEPLOYMENT_NAME = f\"baseline-embedding-{baseline_suffix}\"\n", + "\n", + "print(\"Creating baseline deployment config...\")\n", + "baseline_config = client.inference.deployment_configs.create(\n", + " workspace=\"default\",\n", + " name=BASELINE_DEPLOYMENT_CONFIG,\n", + " engine=\"nim\",\n", + " model_spec={},\n", + " executor_config={\n", + " \"gpu\": 1,\n", + " \"image_name\": NIM_IMAGE,\n", + " \"image_tag\": NIM_TAG,\n", + " },\n", + ")\n", + "\n", + "print(\"Deploying base model...\")\n", + "baseline_deployment = client.inference.deployments.create(\n", + " workspace=\"default\",\n", + " name=BASELINE_DEPLOYMENT_NAME,\n", + " config=baseline_config.name\n", + ")\n", + "print(f\"Baseline deployment: {baseline_deployment.name}\")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "import time\n", + "from IPython.display import clear_output\n", + "\n", + "# Wait for baseline deployment\n", + "TIMEOUT_MINUTES = 15\n", + "start_time = time.time()\n", + "\n", + "print(f\"Waiting for baseline deployment...\")\n", + "while True:\n", + " status = client.inference.deployments.retrieve(\n", + " name=BASELINE_DEPLOYMENT_NAME,\n", + " workspace=\"default\"\n", + " )\n", + " \n", + " elapsed = time.time() - start_time\n", + " elapsed_str = f\"{int(elapsed//60)}m {int(elapsed%60)}s\"\n", + " \n", + " clear_output(wait=True)\n", + " print(f\"Baseline deployment: {status.status} | {elapsed_str}\")\n", + " \n", + " if status.status == \"READY\":\n", + " print(\"Baseline model ready!\")\n", + " if not client.models.wait_for_gateway(BASELINE_DEPLOYMENT_NAME, workspace=\"default\", timeout=60):\n", + " raise RuntimeError(\"Inference gateway did not become ready\")\n", + " break\n", + " if status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n", + " raise RuntimeError(f\"Baseline deployment failed: {status.status}\")\n", + " if elapsed > TIMEOUT_MINUTES * 60:\n", + " raise TimeoutError(\"Baseline deployment timeout\")\n", + " \n", + " time.sleep(10)\n", + "\n" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "# Run baseline ranking with the base model\n", + "BASE_MODEL_ID = \"nvidia/llama-nemotron-embed-1b-v2\"\n", + "\n", + "# Get query embedding\n", + "query_response = client.inference.gateway.provider.post(\n", + " \"v1/embeddings\",\n", + " name=BASELINE_DEPLOYMENT_NAME,\n", + " workspace=\"default\",\n", + " body={\n", + " \"model\": BASE_MODEL_ID,\n", + " \"input\": [DEMO_QUERY],\n", + " \"input_type\": \"query\"\n", + " }\n", + ")\n", + "base_query_emb = query_response[\"data\"][0][\"embedding\"]\n", + "\n", + "# Get document embeddings\n", + "doc_response = client.inference.gateway.provider.post(\n", + " \"v1/embeddings\",\n", + " name=BASELINE_DEPLOYMENT_NAME,\n", + " workspace=\"default\",\n", + " body={\n", + " \"model\": BASE_MODEL_ID,\n", + " \"input\": DEMO_DOCS,\n", + " \"input_type\": \"passage\"\n", + " }\n", + ")\n", + "base_doc_embs = [d[\"embedding\"] for d in doc_response[\"data\"]]\n", + "\n", + "# Calculate similarities and rank\n", + "scores = [(i, cosine_similarity(base_query_emb, base_doc_embs[i])) for i in range(len(DEMO_DOCS))]\n", + "BASELINE_RANKING = sorted(scores, key=lambda x: -x[1])\n", + "\n", + "# Display baseline results\n", + "print(f\"Query: \\\"{DEMO_QUERY}\\\"\\n\")\n", + "print(\"Base Model Ranking:\")\n", + "print(\"-\" * 55)\n", + "for rank, (idx, score) in enumerate(BASELINE_RANKING, 1):\n", + " marker = \" <-- relevant\" if idx in DEMO_RELEVANT else \"\"\n", + " print(f\" #{rank} [{score:.3f}] {DEMO_LABELS[idx]}{marker}\")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "# Delete baseline deployment to free GPU for training\n", + "print(\"Deleting baseline deployment to free GPU...\")\n", + "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace=\"default\")\n", + "\n", + "# Wait for deployment to be fully deleted before deleting config\n", + "print(\"Waiting for deployment deletion...\")\n", + "while True:\n", + " try:\n", + " status = client.inference.deployments.retrieve(name=BASELINE_DEPLOYMENT_NAME, workspace=\"default\")\n", + " if status.status == \"DELETED\":\n", + " break\n", + " print(f\" Status: {status.status}\")\n", + " time.sleep(5)\n", + " except Exception:\n", + " # Deployment no longer exists\n", + " break\n", + "\n", + "# Now safe to delete the config\n", + "client.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace=\"default\")\n", + "print(\"GPU freed. Proceed to fine-tune and improve these rankings.\")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 3. Prepare Dataset\n", + "\n", + "Use the [SPECTER dataset](https://huggingface.co/datasets/embedding-data/SPECTER) from HuggingFace, a collection of scientific paper triplets where papers that cite each other are considered related.\n", + "\n", + "**Dataset structure:**\n", + "- ~684K scientific paper triplets (this tutorial uses 10%)\n", + "- Each triplet: (query paper, positive/related paper, negative/unrelated paper)\n", + "- Papers that cite each other are marked as \"related\"\n", + "\n", + "In this tutorial the following dataset directory structure will be used:\n", + "```\n", + "embedding-dataset\n", + "`-- training.jsonl\n", + "`-- validation.jsonl\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 4. Download and Format SPECTER Dataset\n", + "\n", + "The SPECTER dataset requires conversion to the triplet format required for fine-tuning." + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "from pathlib import Path\n", + "from datasets import load_dataset\n", + "import json\n", + "\n", + "# HuggingFace token for dataset access\n", + "HF_TOKEN = os.environ.get(\"HF_TOKEN\")\n", + "if not HF_TOKEN:\n", + " raise ValueError(\"HF_TOKEN environment variable is required. Get one at https://huggingface.co/settings/tokens\")\n", + "os.environ[\"HF_TOKEN\"] = HF_TOKEN\n", + "\n", + "# Configuration\n", + "DATASET_SIZE = 3000 # Number of triplets (increase for better results, max ~684K)\n", + "VALIDATION_SPLIT = 0.05 # 5% held out for validation\n", + "SEED = 42\n", + "DATASET_PATH = Path(\"embedding-dataset\").absolute()\n", + "\n", + "# Create directory\n", + "os.makedirs(DATASET_PATH, exist_ok=True)\n", + "\n", + "# Download SPECTER dataset\n", + "print(\"Downloading SPECTER dataset...\")\n", + "data = load_dataset(\"embedding-data/SPECTER\")[\"train\"].shuffle(seed=SEED).select(range(DATASET_SIZE))\n", + "\n", + "# Split into train/validation\n", + "print(\"Splitting into train/validation...\")\n", + "splits = data.train_test_split(test_size=VALIDATION_SPLIT, seed=SEED)\n", + "train_data = splits[\"train\"]\n", + "validation_data = splits[\"test\"]\n", + "\n", + "# Convert to triplet JSONL format\n", + "print(\"Saving to JSONL...\")\n", + "for name, dataset in [(\"training\", train_data), (\"validation\", validation_data)]:\n", + " with open(f\"{DATASET_PATH}/{name}.jsonl\", \"w\") as f:\n", + " for row in dataset:\n", + " # SPECTER format: row['set'] = [query, positive, negative]\n", + " triplet = {\n", + " \"query\": row[\"set\"][0],\n", + " \"pos_doc\": row[\"set\"][1],\n", + " \"neg_doc\": [row[\"set\"][2]] # List of negative documents\n", + " }\n", + " f.write(json.dumps(triplet) + \"\\n\")\n", + "\n", + "print(f\"\\nPrepared {len(train_data):,} training, {len(validation_data):,} validation samples\")\n", + "print(f\"\\nExample triplet:\")\n", + "print(f\" Query: {train_data[0]['set'][0][:100]}...\")\n", + "print(f\" Positive: {train_data[0]['set'][1][:100]}...\")\n", + "print(f\" Negative: {train_data[0]['set'][2][:100]}...\")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 5. Create Dataset FileSet and Upload Training Data" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "# Create fileset to store embedding training data\n", + "DATASET_NAME = \"embedding-dataset\"\n", + "\n", + "try:\n", + " client.files.filesets.create(\n", + " workspace=\"default\",\n", + " name=DATASET_NAME,\n", + " description=\"SPECTER embedding training data (scientific paper triplets)\"\n", + " )\n", + " print(f\"Created fileset: {DATASET_NAME}\")\n", + "except ConflictError:\n", + " print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n", + "\n", + "# Upload training data files\n", + "client.files.upload(\n", + " local_path=f\"{DATASET_PATH}/\",\n", + " remote_path=\"\",\n", + " fileset=DATASET_NAME,\n", + " workspace=\"default\"\n", + ")\n", + "\n", + "# Validate upload\n", + "print(\"\\nUploaded files:\")\n", + "print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 6. Secrets Setup\n", + "\n", + "Configure authentication for accessing base models:\n", + "\n", + "- **NGC models** (`ngc://` URIs): Requires NGC API key\n", + "- **HuggingFace models** (`hf://` URIs): Requires HF token for gated/private models\n", + "\n", + "Get your credentials:\n", + "- [NGC API Key](https://ngc.nvidia.com/) (Setup → Generate API Key)\n", + "- [HuggingFace Token](https://huggingface.co/settings/tokens) (Create token with Read access)\n", + "\n", + "---\n", + "\n", + "#### Quick Setup Example\n", + "\n", + "This tutorial fine-tunes [nvidia/llama-3.2-nv-embedqa-1b-v2](https://huggingface.co/nvidia/llama-3.2-nv-embedqa-1b-v2), NVIDIA embedding model optimized for question-answering and retrieval tasks." + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "# Create secrets for model access\n", + "# Note: NGC_API_KEY secret was already created in the baseline step (Step 2)\n", + "HF_TOKEN = os.getenv(\"HF_TOKEN\")\n", + "\n", + "\n", + "def create_or_get_secret(name: str, value: str | None, label: str):\n", + " if not value:\n", + " raise ValueError(f\"{label} is not set\")\n", + " try:\n", + " secret = client.secrets.create(\n", + " name=name,\n", + " workspace=\"default\",\n", + " value=value,\n", + " )\n", + " print(f\"Created secret: {name}\")\n", + " return secret\n", + " except ConflictError:\n", + " print(f\"Secret '{name}' already exists, continuing...\")\n", + " return client.secrets.retrieve(name=name, workspace=\"default\")\n", + "\n", + "\n", + "# Create HuggingFace token secret (for downloading model from HF during training)\n", + "hf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\n", + "print(f\"HF_TOKEN secret: {hf_secret.name}\")\n", + "\n", + "# NGC secret was already created in baseline step (Step 2), or use the platform default\n", + "if \"NGC_SECRET_NAME\" not in globals():\n", + " NGC_SECRET_NAME = \"ngc-api-key\"\n", + "print(f\"NGC_API_KEY secret: {NGC_SECRET_NAME}\")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 7. Create Base Model FileSet and Model Entity\n", + "\n", + "Create a fileset pointing to the [nvidia/llama-3.2-nv-embedqa-1b-v2](https://huggingface.co/nvidia/llama-3.2-nv-embedqa-1b-v2) embedding model from HuggingFace, then create a Model Entity that references this fileset. Model downloading will take place at training time." + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "import time\n", + "from nemo_platform.types.files import HuggingfaceStorageConfigParam\n", + "\n", + "HF_REPO_ID = \"nvidia/llama-nemotron-embed-1b-v2\"\n", + "MODEL_NAME = \"nv-nemotron-embed-1b-base\"\n", + "\n", + "# Ensure you have a HuggingFace token secret created\n", + "try:\n", + " base_model_fs = client.files.filesets.create(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " description=\"NVIDIA Llama 3.2 NV EmbedQA 1B v2 embedding model\",\n", + " storage=HuggingfaceStorageConfigParam(\n", + " type=\"huggingface\",\n", + " # repo_id is the full model name from Hugging Face\n", + " repo_id=HF_REPO_ID,\n", + " repo_type=\"model\",\n", + " # we use the secret created in the previous step\n", + " token_secret=hf_secret.name\n", + " )\n", + " )\n", + "except ConflictError as e:\n", + " print(f\"Base model fileset already exists. Skipping creation.\")\n", + " base_model_fs = client.files.filesets.retrieve(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " )\n", + "\n", + "# Create Model Entity referencing the FileSet\n", + "try:\n", + " base_model = client.models.create(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " fileset=f\"default/{MODEL_NAME}\",\n", + " trust_remote_code=True,\n", + " )\n", + " print(f\"Created Model Entity: {MODEL_NAME}\")\n", + "except ConflictError:\n", + " print(f\"Base model already exists. Updating fileset if different.\")\n", + " base_model = client.models.update(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " fileset=f\"default/{MODEL_NAME}\",\n", + " trust_remote_code=True,\n", + " )\n", + "\n", + "print(f\"\\nBase model fileset: fileset://default/{base_model.name}\")\n", + "print(\"\\nBase model files:\")\n", + "print(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=\"default\").data], indent=2))\n", + "\n", + "# Wait for ModelSpec to be populated from the checkpoint\n", + "print(\"\\nWaiting for ModelSpec to be populated...\")\n", + "SPEC_TIMEOUT_SECONDS = 120\n", + "spec_start = time.time()\n", + "while not base_model.spec:\n", + " if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n", + " raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n", + " time.sleep(2)\n", + " base_model = client.models.retrieve(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " )\n", + "\n", + "print(f\"ModelSpec populated: {base_model.spec}\")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 8. Create Embedding Fine-tuning Job\n", + "\n", + "Create a customization job to fine-tune the embedding model using contrastive learning on the SPECTER dataset.\n", + "\n", + "Submit to the **Automodel** backend using `AutomodelJobInput` with split `schedule`, `batch`, `optimizer`, and `parallelism` sections. Reference the model entity and dataset fileset by workspace/name (not `fileset://` URIs).\n", + "\n", + "**Key hyperparameters for embedding fine-tuning:**\n", + "- **`training.training_type`**: `sft`\n", + "- **`training.finetuning_type`**: `all_weights` for full fine-tuning, or `lora_merged` for merged LoRA\n", + "- **`optimizer.learning_rate`**: Lower values (1e-6 to 5e-6) work well for embedding models\n", + "- **`batch.global_batch_size`**: Larger batches improve contrastive learning (128-256 recommended)\n", + "\n", + "**NOTE:**\n", + "\n", + "NeMo Platform does not support unmerged LoRA adapters for embedding models because the embedding NIM requires ONNX format, which cannot represent standalone adapters. This notebook uses all-weights fine-tuning. For merged LoRA, set `finetuning_type` to `lora_merged`:\n", + "\n", + "```python\n", + "training={\n", + " \"training_type\": \"sft\",\n", + " \"finetuning_type\": \"lora_merged\",\n", + " \"lora\": {\"rank\": 16, \"alpha\": 32},\n", + " \"max_seq_length\": MAX_SEQ_LENGTH,\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "import uuid\n", + "from nemo_automodel_plugin.schema import AutomodelJobInput\n", + "\n", + "job_suffix = uuid.uuid4().hex[:4]\n", + "JOB_NAME = f\"embedding-finetune-job-{job_suffix}\"\n", + "OUTPUT_NAME = f\"nv-embed-finetuned-{job_suffix}\"\n", + "\n", + "EPOCHS = 1\n", + "BATCH_SIZE = 128\n", + "LEARNING_RATE = 5e-6\n", + "MAX_SEQ_LENGTH = 512\n", + "\n", + "spec = AutomodelJobInput(\n", + " model=f\"default/{base_model.name}\",\n", + " dataset={\"training\": f\"default/{DATASET_NAME}\"},\n", + " training={\n", + " \"training_type\": \"sft\",\n", + " \"finetuning_type\": \"all_weights\",\n", + " \"max_seq_length\": MAX_SEQ_LENGTH,\n", + " },\n", + " schedule={\"epochs\": EPOCHS},\n", + " batch={\"global_batch_size\": BATCH_SIZE, \"micro_batch_size\": 1},\n", + " optimizer={\"learning_rate\": LEARNING_RATE},\n", + " parallelism={\"num_gpus_per_node\": 1},\n", + " output={\"name\": OUTPUT_NAME},\n", + ")\n", + "\n", + "job = client.customization.automodel.jobs.create(\n", + " spec=spec, workspace=\"default\", name=JOB_NAME\n", + ")\n", + "\n", + "print(f\"Submitted job: {job.job.name}\")\n", + "print(f\"Output model: {OUTPUT_NAME}\")\n" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 9. Track Training Progress" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "import time\n", + "from IPython.display import clear_output\n", + "\n", + "# Poll job status every 10 seconds until completed\n", + "while True:\n", + " status = client.jobs.get_status(\n", + " name=job.job.name,\n", + " workspace=\"default\"\n", + " )\n", + " \n", + " clear_output(wait=True)\n", + " print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n", + "\n", + " # Extract training progress from nested steps structure\n", + " step: int | None = None\n", + " max_steps: int | None = None\n", + " training_phase: str | None = None\n", + "\n", + " for job_step in status.steps or []:\n", + " if job_step.name == \"training\":\n", + " for task in job_step.tasks or []:\n", + " task_details = task.status_details or {}\n", + " step = task_details.get(\"step\")\n", + " max_steps = task_details.get(\"max_steps\")\n", + " training_phase = task_details.get(\"phase\")\n", + " break\n", + " break\n", + "\n", + " if step is not None and max_steps is not None:\n", + " progress_pct = (step / max_steps) * 100\n", + " print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n", + " if training_phase:\n", + " print(f\"Training Phase: {training_phase}\")\n", + " else:\n", + " print(\"Training step not started yet or progress info not available\")\n", + " \n", + " # Exit loop when job is completed (or failed/cancelled)\n", + " if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n", + " print(f\"\\nJob finished with status: {status.status}\")\n", + " break\n", + " \n", + " time.sleep(10)" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "**Interpreting Embedding Training Metrics:**\n", + "\n", + "Embedding models use contrastive loss—lower values indicate better separation between similar and dissimilar pairs:\n", + "\n", + "| Scenario | Interpretation | Action |\n", + "|----------|----------------|--------|\n", + "| **Loss steadily decreasing** | Model learning semantic relationships | Continue training |\n", + "| **Loss plateaus early** | May need more data or epochs | Increase dataset/epochs |\n", + "| **Loss spikes** | Training instability | Lower learning rate |\n", + "| **Validation loss increasing** | Overfitting | Reduce epochs, add data |" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 10. Deploy Fine-Tuned Embedding Model\n", + "\n", + "After training completes, deploy the embedding model using the Deployment Management Service:" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "# Validate model entity exists\n", + "model_entity = client.models.retrieve(workspace=\"default\", name=OUTPUT_NAME)\n", + "print(model_entity.model_dump_json(indent=2))" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "# Create deployment config for embedding model\n", + "deploy_suffix = uuid.uuid4().hex[:4]\n", + "DEPLOYMENT_CONFIG_NAME = f\"embedding-model-deployment-cfg-{deploy_suffix}\"\n", + "DEPLOYMENT_NAME = f\"embedding-model-deployment-{deploy_suffix}\"\n", + "\n", + "# Embedding NIM image\n", + "NIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\n", + "NIM_TAG = \"1.13.0\" # Update if using newer NIM release\n", + "\n", + "deployment_config = client.inference.deployment_configs.create(\n", + " workspace=\"default\",\n", + " name=DEPLOYMENT_CONFIG_NAME,\n", + " engine=\"nim\",\n", + " model_spec={\n", + " \"model_namespace\": \"default\",\n", + " \"model_name\": OUTPUT_NAME,\n", + " },\n", + " executor_config={\n", + " \"gpu\": 1,\n", + " \"image_name\": NIM_IMAGE,\n", + " \"image_tag\": NIM_TAG,\n", + " },\n", + ")\n", + "\n", + "# Deploy model\n", + "deployment = client.inference.deployments.create(\n", + " workspace=\"default\",\n", + " name=DEPLOYMENT_NAME,\n", + " config=deployment_config.name\n", + ")\n", + "\n", + "print(f\"Deployment name: {deployment.name}\")\n", + "print(f\"Deployment status: {client.inference.deployments.retrieve(name=deployment.name, workspace='default').status}\")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### Track Deployment Status" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "import time\n", + "from IPython.display import clear_output\n", + "\n", + "# Poll deployment status every 15 seconds until ready\n", + "TIMEOUT_MINUTES = 30\n", + "start_time = time.time()\n", + "timeout_seconds = TIMEOUT_MINUTES * 60\n", + "\n", + "print(f\"Monitoring deployment '{deployment.name}'...\")\n", + "print(f\"Timeout: {TIMEOUT_MINUTES} minutes\\n\")\n", + "\n", + "while True:\n", + " deployment_status = client.inference.deployments.retrieve(\n", + " name=deployment.name,\n", + " workspace=\"default\"\n", + " )\n", + " \n", + " elapsed = time.time() - start_time\n", + " elapsed_min = int(elapsed // 60)\n", + " elapsed_sec = int(elapsed % 60)\n", + " \n", + " clear_output(wait=True)\n", + " print(f\"Deployment: {deployment.name}\")\n", + " print(f\"Status: {deployment_status.status}\")\n", + " print(f\"Elapsed time: {elapsed_min}m {elapsed_sec}s\")\n", + " \n", + " # Check if deployment is ready\n", + " if deployment_status.status == \"READY\":\n", + " print(\"\\nDeployment is ready!\")\n", + " if not client.models.wait_for_gateway(deployment.name, workspace=\"default\", timeout=60):\n", + " raise RuntimeError(\"Inference gateway did not become ready\")\n", + " break\n", + " \n", + " # Check for failure states\n", + " if deployment_status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n", + " raise RuntimeError(f\"Deployment failed with status: {deployment_status.status}\")\n", + " \n", + " # Check timeout\n", + " if elapsed > timeout_seconds:\n", + " raise TimeoutError(f\"Deployment timeout after {TIMEOUT_MINUTES} minutes\")\n", + " \n", + " time.sleep(15)" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 11. View the Improvement\n", + "\n", + "Run the same query against the fine-tuned model and compare to the baseline from earlier." + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "# Compare: same query, base model vs fine-tuned\n", + "# Using the same DEMO_QUERY and DEMO_DOCS from the baseline test\n", + "MODEL_ID = f\"default/{OUTPUT_NAME}\"\n", + "\n", + "# Get query embedding from fine-tuned model\n", + "query_response = client.inference.gateway.provider.post(\n", + " \"v1/embeddings\",\n", + " name=deployment.name,\n", + " workspace=\"default\",\n", + " body={\n", + " \"model\": MODEL_ID,\n", + " \"input\": [DEMO_QUERY],\n", + " \"input_type\": \"query\"\n", + " }\n", + ")\n", + "query_embedding = query_response[\"data\"][0][\"embedding\"]\n", + "\n", + "# Get document embeddings from fine-tuned model\n", + "doc_response = client.inference.gateway.provider.post(\n", + " \"v1/embeddings\",\n", + " name=deployment.name,\n", + " workspace=\"default\",\n", + " body={\n", + " \"model\": MODEL_ID,\n", + " \"input\": DEMO_DOCS,\n", + " \"input_type\": \"passage\"\n", + " }\n", + ")\n", + "doc_embeddings = [d[\"embedding\"] for d in doc_response[\"data\"]]\n", + "\n", + "# Calculate similarities and rank\n", + "scores = [(i, cosine_similarity(query_embedding, doc_embeddings[i])) for i in range(len(DEMO_DOCS))]\n", + "FINETUNED_RANKING = sorted(scores, key=lambda x: -x[1])\n", + "\n", + "# Display side-by-side comparison\n", + "print(f\"Query: \\\"{DEMO_QUERY}\\\"\\n\")\n", + "print(f\"{'Rank':<6} {'Base Model':<30} {'Fine-tuned Model':<30}\")\n", + "print(\"-\" * 66)\n", + "\n", + "for rank in range(len(DEMO_DOCS)):\n", + " b_idx, b_score = BASELINE_RANKING[rank]\n", + " f_idx, f_score = FINETUNED_RANKING[rank]\n", + " \n", + " b_label = f\"{DEMO_LABELS[b_idx]} [{b_score:.3f}]\" + (\" *\" if b_idx in DEMO_RELEVANT else \"\")\n", + " f_label = f\"{DEMO_LABELS[f_idx]} [{f_score:.3f}]\" + (\" *\" if f_idx in DEMO_RELEVANT else \"\")\n", + " \n", + " print(f\"#{rank+1:<5} {b_label:<30} {f_label:<30}\")\n", + "\n", + "print(\"\\n* = relevant paper\")\n", + "print(\"\\nThe fine-tuned model pushes 'Random Forest' down and ranks CRF papers higher.\")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### Evaluation Best Practices\n", + "\n", + "**Manual Evaluation** (Recommended)\n", + "- Test with real-world queries from your domain\n", + "- Compare retrieval rankings before and after fine-tuning\n", + "- Check that semantically similar items rank higher than keyword matches\n", + "\n", + "**What to look for:**\n", + "- ✅ Relevant documents consistently rank in top positions\n", + "- ✅ Keyword traps (like \"Random Forest\" vs \"Random Fields\") are handled correctly\n", + "- ✅ Domain-specific terminology is understood\n", + "- ❌ Unrelated documents with matching keywords do not rank high\n", + "\n", + "**Benchmark Evaluation**\n", + "\n", + "For systematic evaluation, use the NeMo Evaluator service with retrieval benchmarks like SciDocs, BEIR, or MTEB. Refer to the [Evaluator documentation](../../evaluator/index.md) for details.\n", + "\n", + "---\n", + "\n", + "## Hyperparameters\n", + "\n", + "For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n", + "\n", + "**Embedding-Specific Recommendations:**\n", + "\n", + "| Parameter | Recommended | Notes |\n", + "|-----------|-------------|-------|\n", + "| `learning_rate` | 1e-6 to 5e-6 | Lower than standard SFT |\n", + "| `batch_size` | 128-256 | Larger batches improve contrastive learning |\n", + "| `max_seq_length` | 512 | Typical for embedding models |\n", + "| `epochs` | 1-3 | Start small, increase if needed |\n", + "\n", + "---\n", + "\n", + "## Troubleshooting\n", + "\n", + "**Embeddings do not show improved retrieval:**\n", + "- Verify dataset quality: triplets should have clear positive/negative distinctions\n", + "- Use hard negatives: negatives should share some overlap with the query but not be relevant (easy negatives do not teach the model much)\n", + "- Increase dataset size: 10K+ triplets recommended for meaningful improvement\n", + "- Try more epochs: embedding models often need multiple passes\n", + "- Lower learning rate: embedding models are sensitive to LR\n", + "\n", + "**Training loss not decreasing:**\n", + "- Check triplet format: ensure `neg_doc` is a list even for single negatives\n", + "- Verify hard negative quality: negatives should be challenging but clearly non-relevant\n", + "- Increase batch size: contrastive learning benefits from larger batches\n", + "\n", + "**Deployment fails:**\n", + "- Ensure you use the correct NIM image for embedding models\n", + "- Verify sufficient GPU memory for the model size\n", + "- Check deployment status: `client.inference.deployments.retrieve(name=deployment.name, workspace=\"default\")` and refer to platform logs for debugging\n", + "\n", + "## Next Steps\n", + "\n", + "- [Monitor training metrics](../manage-customization-jobs/get-job-status.md) in detail\n", + "- [Evaluate your model](../../evaluator/index.md) with retrieval benchmarks\n", + "- Integrate the fine-tuned embedding model into your RAG pipeline\n", + "- Scale up training with the full SPECTER dataset (~684K triplets) for better results" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": ".venv", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.11.14" + } + }, + "nbformat": 4, + "nbformat_minor": 2 +} \ No newline at end of file diff --git a/docs/customizer/tutorials/embedding-customization-job.mdx b/docs/customizer/tutorials/embedding-customization-job.mdx index 8ad23791a8..83bfe8896f 100644 --- a/docs/customizer/tutorials/embedding-customization-job.mdx +++ b/docs/customizer/tutorials/embedding-customization-job.mdx @@ -3,7 +3,802 @@ title: "Embedding Model Customization" description: "" --- - +[Run in Google Colab](https://colab.research.google.com/github/NVIDIA-NeMo/nemo-platform/blob/main/docs/customizer/tutorials/embedding-customization-job.ipynb) + +# Embedding Model Customization + +Learn how to fine-tune an embedding model to improve retrieval accuracy for your specific domain. + +## About + +Embedding models convert text into dense vector representations that capture semantic meaning. Fine-tuning these models on your domain data significantly improves retrieval accuracy—in RAG pipelines, this means the LLM receives more relevant context and produces better answers. + +**What you will achieve with embedding fine-tuning:** + +- 🎯 **Domain specialization:** Adapt general embeddings for legal, medical, scientific, or financial content +- 📈 **Improved retrieval:** Achieve 6-10% better recall on domain-specific benchmarks +- 🔍 **Semantic understanding:** Teach the model your domain's vocabulary and relationships + +**Recall@5** measures the fraction of relevant documents that appear in the top 5 search results. + +**About the baseline:** In retrieval benchmarks like SciDocs, the pretrained model achieves ~0.159 Recall@5. After fine-tuning on scientific paper triplets, you can expect 6-10% improvement (~0.17 Recall@5). + +### Dataset Format for Embedding Models + +Embedding models require **triplet format** for contrastive learning: + +```json +{"query": "What is machine learning?", "pos_doc": "Machine learning is a subset of AI...", "neg_doc": ["Gardening tips for beginners..."]} +``` + +- **`query`**: The search query or question +- **`pos_doc`**: A document relevant to the query (positive example) +- **`neg_doc`**: List of hard negatives—documents that share some overlap with the query but are not actually relevant (negative example) + +The model learns to maximize similarity between query and positive document while minimizing similarity with negative documents. + +## Prerequisites + +Before starting this tutorial, ensure you have: + +1. **Completed the [Quickstart](/documentation/get-started)** to install and deploy NeMo Platform locally +2. **Installed the Python SDK** (PyPI wrapper: `pip install "nemo-platform[all]"`; source checkout: run `make bootstrap` from the repository root) +3. **HuggingFace token** with read access to download the SPECTER dataset (get one at [huggingface.co/settings/tokens](https://huggingface.co/settings/tokens)) +4. **NGC API key** to pull NIM container images from nvcr.io (get one at [ngc.nvidia.com](https://ngc.nvidia.com/) → Setup → Generate API Key) + +## Quick Start + +### 1. Initialize SDK + +The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](/documentation/get-started) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable: + +```sh +export NMP_BASE_URL= +``` + +```python +import json +import os +from nemo_platform import NeMoPlatform, ConflictError + +NMP_BASE_URL = os.environ.get("NMP_BASE_URL", "http://localhost:8080") +client = NeMoPlatform( + base_url=NMP_BASE_URL, + workspace="default" +) +``` + +### 2. Establish Baseline Performance + +Before fine-tuning, establish baseline performance with the pretrained model. Deploy it, run a test query, and observe where it struggles. Following fine-tuning, compare the results. + +**Scenario:** Searching scientific papers by meaning, not keywords. + +**Demo setup:** +- **Query:** "Conditional Random Fields" (CRFs) - a method for sequence labeling in NLP +- **Trap:** "Random Forests" shares the word "random" but is an unrelated tree-based algorithm +- **Goal:** The model should distinguish between them. + +```python +# Install required packages for dataset preparation +%pip install -q datasets huggingface_hub +``` + +```python +import uuid +import numpy as np + +# Demo query and documents for baseline comparison +DEMO_QUERY = "Conditional Random Fields: Probabilistic Models for Segmenting and Labeling Sequence Data" + +DEMO_DOCS = [ + "Bidirectional LSTM-CRF Models for Sequence Tagging", # CRF-based paper + "An Introduction to Conditional Random Fields", # CRF tutorial + "Random Forests", # Keyword trap! Unrelated. + "Neural Architectures for Named Entity Recognition", # Related to sequence labeling; may use CRFs + "Support Vector Machines for Classification", # Unrelated ML method +] + +DEMO_LABELS = ["BiLSTM-CRF", "CRF Tutorial", "Random Forest", "NER", "SVM"] +DEMO_RELEVANT = {0, 1, 3} # Papers actually relevant to CRFs + +def cosine_similarity(a, b): + """Calculate cosine similarity between two vectors.""" + return np.dot(a, b) / (np.linalg.norm(a) * np.linalg.norm(b)) +``` + +```python +# NGC API key is required to pull NIM images from nvcr.io +NGC_API_KEY = os.environ.get("NGC_API_KEY") +if not NGC_API_KEY: + raise ValueError("NGC_API_KEY environment variable is required. Get one at https://ngc.nvidia.com/ → Setup → Generate API Key") + +# Create NGC secret for pulling NIM images +NGC_SECRET_NAME = "ngc-api-key" +try: + client.secrets.create(name=NGC_SECRET_NAME, workspace="default", value=NGC_API_KEY) + print(f"Created secret: {NGC_SECRET_NAME}") +except ConflictError: + print(f"Secret '{NGC_SECRET_NAME}' already exists, continuing...") + +# Deploy base model for baseline comparison +BASE_MODEL_HF = "nvidia/llama-nemotron-embed-1b-v2" +NIM_IMAGE = "nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2" +NIM_TAG = "1.13.0" + +baseline_suffix = uuid.uuid4().hex[:4] +BASELINE_DEPLOYMENT_CONFIG = f"baseline-embedding-cfg-{baseline_suffix}" +BASELINE_DEPLOYMENT_NAME = f"baseline-embedding-{baseline_suffix}" + +print("Creating baseline deployment config...") +baseline_config = client.inference.deployment_configs.create( + workspace="default", + name=BASELINE_DEPLOYMENT_CONFIG, + engine="nim", + model_spec={}, + executor_config={ + "gpu": 1, + "image_name": NIM_IMAGE, + "image_tag": NIM_TAG, + }, +) + +print("Deploying base model...") +baseline_deployment = client.inference.deployments.create( + workspace="default", + name=BASELINE_DEPLOYMENT_NAME, + config=baseline_config.name +) +print(f"Baseline deployment: {baseline_deployment.name}") +``` + +```python +import time +from IPython.display import clear_output + +# Wait for baseline deployment +TIMEOUT_MINUTES = 15 +start_time = time.time() + +print(f"Waiting for baseline deployment...") +while True: + status = client.inference.deployments.retrieve( + name=BASELINE_DEPLOYMENT_NAME, + workspace="default" + ) + + elapsed = time.time() - start_time + elapsed_str = f"{int(elapsed//60)}m {int(elapsed%60)}s" + + clear_output(wait=True) + print(f"Baseline deployment: {status.status} | {elapsed_str}") + + if status.status == "READY": + print("Baseline model ready!") + if not client.models.wait_for_gateway(BASELINE_DEPLOYMENT_NAME, workspace="default", timeout=60): + raise RuntimeError("Inference gateway did not become ready") + break + if status.status in ("FAILED", "ERROR", "TERMINATED", "LOST"): + raise RuntimeError(f"Baseline deployment failed: {status.status}") + if elapsed > TIMEOUT_MINUTES * 60: + raise TimeoutError("Baseline deployment timeout") + + time.sleep(10) + + +``` + +```python +# Run baseline ranking with the base model +BASE_MODEL_ID = "nvidia/llama-nemotron-embed-1b-v2" + +# Get query embedding +query_response = client.inference.gateway.provider.post( + "v1/embeddings", + name=BASELINE_DEPLOYMENT_NAME, + workspace="default", + body={ + "model": BASE_MODEL_ID, + "input": [DEMO_QUERY], + "input_type": "query" + } +) +base_query_emb = query_response["data"][0]["embedding"] + +# Get document embeddings +doc_response = client.inference.gateway.provider.post( + "v1/embeddings", + name=BASELINE_DEPLOYMENT_NAME, + workspace="default", + body={ + "model": BASE_MODEL_ID, + "input": DEMO_DOCS, + "input_type": "passage" + } +) +base_doc_embs = [d["embedding"] for d in doc_response["data"]] + +# Calculate similarities and rank +scores = [(i, cosine_similarity(base_query_emb, base_doc_embs[i])) for i in range(len(DEMO_DOCS))] +BASELINE_RANKING = sorted(scores, key=lambda x: -x[1]) + +# Display baseline results +print(f"Query: \"{DEMO_QUERY}\"\n") +print("Base Model Ranking:") +print("-" * 55) +for rank, (idx, score) in enumerate(BASELINE_RANKING, 1): + marker = " <-- relevant" if idx in DEMO_RELEVANT else "" + print(f" #{rank} [{score:.3f}] {DEMO_LABELS[idx]}{marker}") +``` + +```python +# Delete baseline deployment to free GPU for training +print("Deleting baseline deployment to free GPU...") +client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace="default") + +# Wait for deployment to be fully deleted before deleting config +print("Waiting for deployment deletion...") +while True: + try: + status = client.inference.deployments.retrieve(name=BASELINE_DEPLOYMENT_NAME, workspace="default") + if status.status == "DELETED": + break + print(f" Status: {status.status}") + time.sleep(5) + except Exception: + # Deployment no longer exists + break + +# Now safe to delete the config +client.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace="default") +print("GPU freed. Proceed to fine-tune and improve these rankings.") +``` + +### 3. Prepare Dataset + +Use the [SPECTER dataset](https://huggingface.co/datasets/embedding-data/SPECTER) from HuggingFace, a collection of scientific paper triplets where papers that cite each other are considered related. + +**Dataset structure:** +- ~684K scientific paper triplets (this tutorial uses 10%) +- Each triplet: (query paper, positive/related paper, negative/unrelated paper) +- Papers that cite each other are marked as "related" + +In this tutorial the following dataset directory structure will be used: +``` +embedding-dataset +`-- training.jsonl +`-- validation.jsonl +``` + +### 4. Download and Format SPECTER Dataset + +The SPECTER dataset requires conversion to the triplet format required for fine-tuning. + +```python +from pathlib import Path +from datasets import load_dataset +import json + +# HuggingFace token for dataset access +HF_TOKEN = os.environ.get("HF_TOKEN") +if not HF_TOKEN: + raise ValueError("HF_TOKEN environment variable is required. Get one at https://huggingface.co/settings/tokens") +os.environ["HF_TOKEN"] = HF_TOKEN + +# Configuration +DATASET_SIZE = 3000 # Number of triplets (increase for better results, max ~684K) +VALIDATION_SPLIT = 0.05 # 5% held out for validation +SEED = 42 +DATASET_PATH = Path("embedding-dataset").absolute() + +# Create directory +os.makedirs(DATASET_PATH, exist_ok=True) + +# Download SPECTER dataset +print("Downloading SPECTER dataset...") +data = load_dataset("embedding-data/SPECTER")["train"].shuffle(seed=SEED).select(range(DATASET_SIZE)) + +# Split into train/validation +print("Splitting into train/validation...") +splits = data.train_test_split(test_size=VALIDATION_SPLIT, seed=SEED) +train_data = splits["train"] +validation_data = splits["test"] + +# Convert to triplet JSONL format +print("Saving to JSONL...") +for name, dataset in [("training", train_data), ("validation", validation_data)]: + with open(f"{DATASET_PATH}/{name}.jsonl", "w") as f: + for row in dataset: + # SPECTER format: row['set'] = [query, positive, negative] + triplet = { + "query": row["set"][0], + "pos_doc": row["set"][1], + "neg_doc": [row["set"][2]] # List of negative documents + } + f.write(json.dumps(triplet) + "\n") + +print(f"\nPrepared {len(train_data):,} training, {len(validation_data):,} validation samples") +print(f"\nExample triplet:") +print(f" Query: {train_data[0]['set'][0][:100]}...") +print(f" Positive: {train_data[0]['set'][1][:100]}...") +print(f" Negative: {train_data[0]['set'][2][:100]}...") +``` + +### 5. Create Dataset FileSet and Upload Training Data + +```python +# Create fileset to store embedding training data +DATASET_NAME = "embedding-dataset" + +try: + client.files.filesets.create( + workspace="default", + name=DATASET_NAME, + description="SPECTER embedding training data (scientific paper triplets)" + ) + print(f"Created fileset: {DATASET_NAME}") +except ConflictError: + print(f"Fileset '{DATASET_NAME}' already exists, continuing...") + +# Upload training data files +client.files.upload( + local_path=f"{DATASET_PATH}/", + remote_path="", + fileset=DATASET_NAME, + workspace="default" +) + +# Validate upload +print("\nUploaded files:") +print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2)) +``` + +### 6. Secrets Setup + +Configure authentication for accessing base models: + +- **NGC models** (`ngc://` URIs): Requires NGC API key +- **HuggingFace models** (`hf://` URIs): Requires HF token for gated/private models + +Get your credentials: +- [NGC API Key](https://ngc.nvidia.com/) (Setup → Generate API Key) +- [HuggingFace Token](https://huggingface.co/settings/tokens) (Create token with Read access) + +--- + +#### Quick Setup Example + +This tutorial fine-tunes [nvidia/llama-3.2-nv-embedqa-1b-v2](https://huggingface.co/nvidia/llama-3.2-nv-embedqa-1b-v2), NVIDIA embedding model optimized for question-answering and retrieval tasks. + +```python +# Create secrets for model access +# Note: NGC_API_KEY secret was already created in the baseline step (Step 2) +HF_TOKEN = os.getenv("HF_TOKEN") + + +def create_or_get_secret(name: str, value: str | None, label: str): + if not value: + raise ValueError(f"{label} is not set") + try: + secret = client.secrets.create( + name=name, + workspace="default", + value=value, + ) + print(f"Created secret: {name}") + return secret + except ConflictError: + print(f"Secret '{name}' already exists, continuing...") + return client.secrets.retrieve(name=name, workspace="default") + + +# Create HuggingFace token secret (for downloading model from HF during training) +hf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN") +print(f"HF_TOKEN secret: {hf_secret.name}") + +# NGC secret was already created in baseline step (Step 2), or use the platform default +if "NGC_SECRET_NAME" not in globals(): + NGC_SECRET_NAME = "ngc-api-key" +print(f"NGC_API_KEY secret: {NGC_SECRET_NAME}") +``` + +### 7. Create Base Model FileSet and Model Entity + +Create a fileset pointing to the [nvidia/llama-3.2-nv-embedqa-1b-v2](https://huggingface.co/nvidia/llama-3.2-nv-embedqa-1b-v2) embedding model from HuggingFace, then create a Model Entity that references this fileset. Model downloading will take place at training time. + +```python +import time +from nemo_platform.types.files import HuggingfaceStorageConfigParam + +HF_REPO_ID = "nvidia/llama-nemotron-embed-1b-v2" +MODEL_NAME = "nv-nemotron-embed-1b-base" + +# Ensure you have a HuggingFace token secret created +try: + base_model_fs = client.files.filesets.create( + workspace="default", + name=MODEL_NAME, + description="NVIDIA Llama 3.2 NV EmbedQA 1B v2 embedding model", + storage=HuggingfaceStorageConfigParam( + type="huggingface", + # repo_id is the full model name from Hugging Face + repo_id=HF_REPO_ID, + repo_type="model", + # we use the secret created in the previous step + token_secret=hf_secret.name + ) + ) +except ConflictError as e: + print(f"Base model fileset already exists. Skipping creation.") + base_model_fs = client.files.filesets.retrieve( + workspace="default", + name=MODEL_NAME, + ) + +# Create Model Entity referencing the FileSet +try: + base_model = client.models.create( + workspace="default", + name=MODEL_NAME, + fileset=f"default/{MODEL_NAME}", + trust_remote_code=True, + ) + print(f"Created Model Entity: {MODEL_NAME}") +except ConflictError: + print(f"Base model already exists. Updating fileset if different.") + base_model = client.models.update( + workspace="default", + name=MODEL_NAME, + fileset=f"default/{MODEL_NAME}", + trust_remote_code=True, + ) + +print(f"\nBase model fileset: fileset://default/{base_model.name}") +print("\nBase model files:") +print(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace="default").data], indent=2)) + +# Wait for ModelSpec to be populated from the checkpoint +print("\nWaiting for ModelSpec to be populated...") +SPEC_TIMEOUT_SECONDS = 120 +spec_start = time.time() +while not base_model.spec: + if time.time() - spec_start > SPEC_TIMEOUT_SECONDS: + raise TimeoutError(f"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds") + time.sleep(2) + base_model = client.models.retrieve( + workspace="default", + name=MODEL_NAME, + ) + +print(f"ModelSpec populated: {base_model.spec}") +``` + +### 8. Create Embedding Fine-tuning Job + +Create a customization job to fine-tune the embedding model using contrastive learning on the SPECTER dataset. + +Submit to the **Automodel** backend using `AutomodelJobInput` with split `schedule`, `batch`, `optimizer`, and `parallelism` sections. Reference the model entity and dataset fileset by workspace/name (not `fileset://` URIs). + +**Key hyperparameters for embedding fine-tuning:** +- **`training.training_type`**: `sft` +- **`training.finetuning_type`**: `all_weights` for full fine-tuning, or `lora_merged` for merged LoRA +- **`optimizer.learning_rate`**: Lower values (1e-6 to 5e-6) work well for embedding models +- **`batch.global_batch_size`**: Larger batches improve contrastive learning (128-256 recommended) + +**NOTE:** + +NeMo Platform does not support unmerged LoRA adapters for embedding models because the embedding NIM requires ONNX format, which cannot represent standalone adapters. This notebook uses all-weights fine-tuning. For merged LoRA, set `finetuning_type` to `lora_merged`: + +```python +training={ + "training_type": "sft", + "finetuning_type": "lora_merged", + "lora": {"rank": 16, "alpha": 32}, + "max_seq_length": MAX_SEQ_LENGTH, +} +``` + +```python +import uuid +from nemo_automodel_plugin.schema import AutomodelJobInput + +job_suffix = uuid.uuid4().hex[:4] +JOB_NAME = f"embedding-finetune-job-{job_suffix}" +OUTPUT_NAME = f"nv-embed-finetuned-{job_suffix}" + +EPOCHS = 1 +BATCH_SIZE = 128 +LEARNING_RATE = 5e-6 +MAX_SEQ_LENGTH = 512 + +spec = AutomodelJobInput( + model=f"default/{base_model.name}", + dataset={"training": f"default/{DATASET_NAME}"}, + training={ + "training_type": "sft", + "finetuning_type": "all_weights", + "max_seq_length": MAX_SEQ_LENGTH, + }, + schedule={"epochs": EPOCHS}, + batch={"global_batch_size": BATCH_SIZE, "micro_batch_size": 1}, + optimizer={"learning_rate": LEARNING_RATE}, + parallelism={"num_gpus_per_node": 1}, + output={"name": OUTPUT_NAME}, +) + +job = client.customization.automodel.jobs.create( + spec=spec, workspace="default", name=JOB_NAME +) + +print(f"Submitted job: {job.job.name}") +print(f"Output model: {OUTPUT_NAME}") + +``` + +### 9. Track Training Progress + +```python +import time +from IPython.display import clear_output + +# Poll job status every 10 seconds until completed +while True: + status = client.jobs.get_status( + name=job.job.name, + workspace="default" + ) + + clear_output(wait=True) + print(f"Job Status: {status.model_dump_json(indent=2)}") + + # Extract training progress from nested steps structure + step: int | None = None + max_steps: int | None = None + training_phase: str | None = None + + for job_step in status.steps or []: + if job_step.name == "training": + for task in job_step.tasks or []: + task_details = task.status_details or {} + step = task_details.get("step") + max_steps = task_details.get("max_steps") + training_phase = task_details.get("phase") + break + break + + if step is not None and max_steps is not None: + progress_pct = (step / max_steps) * 100 + print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)") + if training_phase: + print(f"Training Phase: {training_phase}") + else: + print("Training step not started yet or progress info not available") + + # Exit loop when job is completed (or failed/cancelled) + if status.status in ("completed", "failed", "cancelled", "error"): + print(f"\nJob finished with status: {status.status}") + break + + time.sleep(10) +``` + +**Interpreting Embedding Training Metrics:** + +Embedding models use contrastive loss—lower values indicate better separation between similar and dissimilar pairs: + +| Scenario | Interpretation | Action | +|----------|----------------|--------| +| **Loss steadily decreasing** | Model learning semantic relationships | Continue training | +| **Loss plateaus early** | May need more data or epochs | Increase dataset/epochs | +| **Loss spikes** | Training instability | Lower learning rate | +| **Validation loss increasing** | Overfitting | Reduce epochs, add data | + +### 10. Deploy Fine-Tuned Embedding Model + +After training completes, deploy the embedding model using the Deployment Management Service: + +```python +# Validate model entity exists +model_entity = client.models.retrieve(workspace="default", name=OUTPUT_NAME) +print(model_entity.model_dump_json(indent=2)) +``` + +```python +# Create deployment config for embedding model +deploy_suffix = uuid.uuid4().hex[:4] +DEPLOYMENT_CONFIG_NAME = f"embedding-model-deployment-cfg-{deploy_suffix}" +DEPLOYMENT_NAME = f"embedding-model-deployment-{deploy_suffix}" + +# Embedding NIM image +NIM_IMAGE = "nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2" +NIM_TAG = "1.13.0" # Update if using newer NIM release + +deployment_config = client.inference.deployment_configs.create( + workspace="default", + name=DEPLOYMENT_CONFIG_NAME, + engine="nim", + model_spec={ + "model_namespace": "default", + "model_name": OUTPUT_NAME, + }, + executor_config={ + "gpu": 1, + "image_name": NIM_IMAGE, + "image_tag": NIM_TAG, + }, +) + +# Deploy model +deployment = client.inference.deployments.create( + workspace="default", + name=DEPLOYMENT_NAME, + config=deployment_config.name +) + +print(f"Deployment name: {deployment.name}") +print(f"Deployment status: {client.inference.deployments.retrieve(name=deployment.name, workspace='default').status}") +``` + +#### Track Deployment Status + +```python +import time +from IPython.display import clear_output + +# Poll deployment status every 15 seconds until ready +TIMEOUT_MINUTES = 30 +start_time = time.time() +timeout_seconds = TIMEOUT_MINUTES * 60 + +print(f"Monitoring deployment '{deployment.name}'...") +print(f"Timeout: {TIMEOUT_MINUTES} minutes\n") + +while True: + deployment_status = client.inference.deployments.retrieve( + name=deployment.name, + workspace="default" + ) + + elapsed = time.time() - start_time + elapsed_min = int(elapsed // 60) + elapsed_sec = int(elapsed % 60) + + clear_output(wait=True) + print(f"Deployment: {deployment.name}") + print(f"Status: {deployment_status.status}") + print(f"Elapsed time: {elapsed_min}m {elapsed_sec}s") + + # Check if deployment is ready + if deployment_status.status == "READY": + print("\nDeployment is ready!") + if not client.models.wait_for_gateway(deployment.name, workspace="default", timeout=60): + raise RuntimeError("Inference gateway did not become ready") + break + + # Check for failure states + if deployment_status.status in ("FAILED", "ERROR", "TERMINATED", "LOST"): + raise RuntimeError(f"Deployment failed with status: {deployment_status.status}") + + # Check timeout + if elapsed > timeout_seconds: + raise TimeoutError(f"Deployment timeout after {TIMEOUT_MINUTES} minutes") + + time.sleep(15) +``` + +### 11. View the Improvement + +Run the same query against the fine-tuned model and compare to the baseline from earlier. + +```python +# Compare: same query, base model vs fine-tuned +# Using the same DEMO_QUERY and DEMO_DOCS from the baseline test +MODEL_ID = f"default/{OUTPUT_NAME}" + +# Get query embedding from fine-tuned model +query_response = client.inference.gateway.provider.post( + "v1/embeddings", + name=deployment.name, + workspace="default", + body={ + "model": MODEL_ID, + "input": [DEMO_QUERY], + "input_type": "query" + } +) +query_embedding = query_response["data"][0]["embedding"] + +# Get document embeddings from fine-tuned model +doc_response = client.inference.gateway.provider.post( + "v1/embeddings", + name=deployment.name, + workspace="default", + body={ + "model": MODEL_ID, + "input": DEMO_DOCS, + "input_type": "passage" + } +) +doc_embeddings = [d["embedding"] for d in doc_response["data"]] + +# Calculate similarities and rank +scores = [(i, cosine_similarity(query_embedding, doc_embeddings[i])) for i in range(len(DEMO_DOCS))] +FINETUNED_RANKING = sorted(scores, key=lambda x: -x[1]) + +# Display side-by-side comparison +print(f"Query: \"{DEMO_QUERY}\"\n") +print(f"{'Rank':<6} {'Base Model':<30} {'Fine-tuned Model':<30}") +print("-" * 66) + +for rank in range(len(DEMO_DOCS)): + b_idx, b_score = BASELINE_RANKING[rank] + f_idx, f_score = FINETUNED_RANKING[rank] + + b_label = f"{DEMO_LABELS[b_idx]} [{b_score:.3f}]" + (" *" if b_idx in DEMO_RELEVANT else "") + f_label = f"{DEMO_LABELS[f_idx]} [{f_score:.3f}]" + (" *" if f_idx in DEMO_RELEVANT else "") + + print(f"#{rank+1:<5} {b_label:<30} {f_label:<30}") + +print("\n* = relevant paper") +print("\nThe fine-tuned model pushes 'Random Forest' down and ranks CRF papers higher.") +``` + +### Evaluation Best Practices + +**Manual Evaluation** (Recommended) +- Test with real-world queries from your domain +- Compare retrieval rankings before and after fine-tuning +- Check that semantically similar items rank higher than keyword matches + +**What to look for:** +- ✅ Relevant documents consistently rank in top positions +- ✅ Keyword traps (like "Random Forest" vs "Random Fields") are handled correctly +- ✅ Domain-specific terminology is understood +- ❌ Unrelated documents with matching keywords do not rank high + +**Benchmark Evaluation** + +For systematic evaluation, use the NeMo Evaluator service with retrieval benchmarks like SciDocs, BEIR, or MTEB. Refer to the [Evaluator documentation](/documentation/evaluate-models) for details. + +--- + +## Hyperparameters + +For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](/documentation/customizer-reference/manage-customization-jobs/training-configuration). + +**Embedding-Specific Recommendations:** + +| Parameter | Recommended | Notes | +|-----------|-------------|-------| +| `learning_rate` | 1e-6 to 5e-6 | Lower than standard SFT | +| `batch_size` | 128-256 | Larger batches improve contrastive learning | +| `max_seq_length` | 512 | Typical for embedding models | +| `epochs` | 1-3 | Start small, increase if needed | + +--- + +## Troubleshooting + +**Embeddings do not show improved retrieval:** +- Verify dataset quality: triplets should have clear positive/negative distinctions +- Use hard negatives: negatives should share some overlap with the query but not be relevant (easy negatives do not teach the model much) +- Increase dataset size: 10K+ triplets recommended for meaningful improvement +- Try more epochs: embedding models often need multiple passes +- Lower learning rate: embedding models are sensitive to LR + +**Training loss not decreasing:** +- Check triplet format: ensure `neg_doc` is a list even for single negatives +- Verify hard negative quality: negatives should be challenging but clearly non-relevant +- Increase batch size: contrastive learning benefits from larger batches + +**Deployment fails:** +- Ensure you use the correct NIM image for embedding models +- Verify sufficient GPU memory for the model size +- Check deployment status: `client.inference.deployments.retrieve(name=deployment.name, workspace="default")` and refer to platform logs for debugging + +## Next Steps + +- [Monitor training metrics](/documentation/customizer-reference/manage-customization-jobs/get-job-status) in detail +- [Evaluate your model](/documentation/evaluate-models) with retrieval benchmarks +- Integrate the fine-tuned embedding model into your RAG pipeline +- Scale up training with the full SPECTER dataset (~684K triplets) for better results diff --git a/docs/customizer/tutorials/format-training-dataset.mdx b/docs/customizer/tutorials/format-training-dataset.mdx index 43e85fef50..4217a47763 100644 --- a/docs/customizer/tutorials/format-training-dataset.mdx +++ b/docs/customizer/tutorials/format-training-dataset.mdx @@ -2,6 +2,7 @@ title: "Format Training Dataset" description: "" --- + Learn how to format a training dataset to work with the model type you want to train, such as a **chat** or **completion** model. @@ -40,32 +41,36 @@ Before formatting your dataset, follow these principles to ensure high-quality t Before uploading your dataset, verify: -✅ **Format:** All entries are valid JSON on a single line (JSONL) -✅ **Schema:** Required fields present (`messages`, `prompt`/`completion`, or custom columns) -✅ **Encoding:** UTF-8 encoding (not UTF-16 or other encodings) -✅ **Completeness:** No empty or null values in required fields -✅ **Length:** Examples fit within model's context window (typically 2048-8192 tokens) +✅ **Format:** All entries are valid JSON on a single line (JSONL) +✅ **Schema:** Required fields present (`messages`, `prompt`/`completion`, or custom columns) +✅ **Encoding:** UTF-8 encoding (not UTF-16 or other encodings) +✅ **Completeness:** No empty or null values in required fields +✅ **Length:** Examples fit within model's context window (typically 2048-8192 tokens) ✅ **Split:** Training and validation sets are separate and representative ### Common Dataset Issues **Overfitting symptoms:** + - Training loss → 0 but validation loss stays high or increases - Model memorizes training examples verbatim - Poor generalization to new inputs **Solutions:** + - Increase dataset size and diversity - Reduce training epochs (try 1-3 instead of 5+) - Add more validation data - Use regularization (dropout, lower learning rate) **Underfitting symptoms:** + - Both training and validation loss remain high - Model output quality is poor even on training examples - Loss plateaus early in training **Solutions:** + - Increase training epochs (try 5-10 instead of 3) - Increase model size if using small models - Check data quality and consistency @@ -103,10 +108,12 @@ For illustrative purposes only, we show an example entry as multi-line JSON. { "role": "system", "content": "" - }, { + }, + { "role": "user", "content": "" - }, { + }, + { "role": "assistant", "content": "" } @@ -135,22 +142,29 @@ For example, if your original system message is `You are a helpful assistant.`, ```json -{"messages": [ - {"role": "system", "content": "detailed thinking off"}, - {"role": "user", "content": "What is 2 + 2?"}, - {"role": "assistant", "content": "4"} -]} +{ + "messages": [ + { "role": "system", "content": "detailed thinking off" }, + { "role": "user", "content": "What is 2 + 2?" }, + { "role": "assistant", "content": "4" } + ] +} ``` ```json -{"messages": [ - {"role": "system", "content": "detailed thinking on"}, - {"role": "user", "content": "What is 2 + 2?"}, - {"role": "assistant", "content": "To solve 2 + 2, add 2 and 2 together. The answer is 4."} -]} +{ + "messages": [ + { "role": "system", "content": "detailed thinking on" }, + { "role": "user", "content": "What is 2 + 2?" }, + { + "role": "assistant", + "content": "To solve 2 + 2, add 2 and 2 together. The answer is 4." + } + ] +} ``` @@ -187,41 +201,46 @@ For illustrative purposes only, we show an example entry as multi-line JSON. { "role": "assistant", "content": "", - "tool_calls": [{ - "type": "function", - "function": { - "name": "fibonacci", - "arguments": {"n": 20} + "tool_calls": [ + { + "type": "function", + "function": { + "name": "fibonacci", + "arguments": { "n": 20 } + } } - }] + ] } ], - "tools": [{ - "type": "function", - "function": { - "name": "fibonacci", - "description": "Calculates the nth Fibonacci number.", - "parameters": { - "type": "object", - "properties": { - "n": { - "description": "The position of the Fibonacci number.", - "type": "integer" + "tools": [ + { + "type": "function", + "function": { + "name": "fibonacci", + "description": "Calculates the nth Fibonacci number.", + "parameters": { + "type": "object", + "properties": { + "n": { + "description": "The position of the Fibonacci number.", + "type": "integer" + } } } } } - }] + ] } ``` ##### Shared Tools -When your dataset uses the same set of tools across all examples, you can streamline your configuration by omitting the `tools` field from individual dataset entries. Instead, specify these tools once in the `dataset_parameters` section during job creation. +When your dataset uses the same set of tools across all examples, include the shared tool definitions in **each training row** (Automodel jobs do not accept a separate `dataset_parameters` block). See the [Tool Calling tutorial](/documentation/example-applications/tool-calling) for a full workflow. ```python import os +from nemo_automodel_plugin.schema import AutomodelJobInput from nemo_platform import NeMoPlatform client = NeMoPlatform( @@ -229,44 +248,23 @@ client = NeMoPlatform( workspace="default", ) -# Create a customization job with shared tools -job = client.customization.jobs.create( +# Create an Automodel job (include tools in each JSONL training example) +spec = AutomodelJobInput( + model="default/llama-3-2-1b", + dataset={"training": "default/my-tool-dataset"}, + training={"training_type": "sft", "finetuning_type": "lora", "lora": {"rank": 8, "alpha": 32}}, + schedule={"epochs": 10}, + batch={"global_batch_size": 16, "micro_batch_size": 1}, + optimizer={"learning_rate": 1e-4}, +) + +job = client.customization.automodel.jobs.create( name="my-tool-calling-job", workspace="default", - spec={ - "model": "default/llama-3-2-1b", - "dataset": "fileset://default/my-tool-dataset", - "dataset_parameters": { - "tools": [ - { - "type": "function", - "function": { - "name": "fibonacci", - "description": "Calculates the nth Fibonacci number.", - "parameters": { - "type": "object", - "properties": { - "n": { - "description": "The position of the Fibonacci number.", - "type": "integer", - } - }, - }, - }, - } - ] - }, - "training": { - "type": "sft", - "peft": {"type": "lora", "rank": 8, "alpha": 32}, - "epochs": 10, - "batch_size": 16, - "learning_rate": 1e-4, - }, - }, + spec=spec, ) -print(f"Created job: {job.name}") +print(f"Submitted job: {job.job.name}") ``` ### Find Chat Models @@ -307,27 +305,30 @@ import os from nemo_platform import NeMoPlatform client = NeMoPlatform( - base_url=os.environ.get("NMP_BASE_URL", "http://localhost:8080"), - workspace="default", +base_url=os.environ.get("NMP_BASE_URL", "http://localhost:8080"), +workspace="default", ) # Use Inference Gateway with OpenAI-compatible client + oai_client = client.models.get_openai_client(workspace="default") # Create a chat completion (model format: "workspace/model-entity-name") + response = oai_client.chat.completions.create( - model="default/llama-3-2-1b", - messages=[ - {"role": "system", "content": "You are a helpful assistant."}, - {"role": "user", "content": "Hello! How are you?"}, - ], - temperature=0.5, - top_p=1, - max_tokens=1024, +model="default/llama-3-2-1b", +messages=[ +{"role": "system", "content": "You are a helpful assistant."}, +{"role": "user", "content": "Hello! How are you?"}, +], +temperature=0.5, +top_p=1, +max_tokens=1024, ) print(f"Response: {response.choices[0].message.content}") -``` + +```` ## Next Steps @@ -380,4 +381,4 @@ response = oai_client.completions.create( ) print(f"Response: {response.choices[0].text}") -``` +```` diff --git a/docs/customizer/tutorials/import-hf-model.mdx b/docs/customizer/tutorials/import-hf-model.mdx index fdfff6b8ee..6f89cbddf5 100644 --- a/docs/customizer/tutorials/import-hf-model.mdx +++ b/docs/customizer/tutorials/import-hf-model.mdx @@ -577,7 +577,7 @@ If you included a WandB API key, you can view your training results at [wandb.ai base_model_id = f"{NAMESPACE}/{MODEL_NAME}" # Option 1: If you still have the job object from creation -# lora_model_id = job.spec.output.name +# lora_model_id = OUTPUT_NAME # Option 2: Construct from job parameters (use this if running in a new session) lora_model_id = f"{NAMESPACE}/{MODEL_NAME}-lora@v{MODEL_VERSION}" diff --git a/docs/customizer/tutorials/lora-customization-job.ipynb b/docs/customizer/tutorials/lora-customization-job.ipynb index 39a34bf30e..6bb42e9947 100644 --- a/docs/customizer/tutorials/lora-customization-job.ipynb +++ b/docs/customizer/tutorials/lora-customization-job.ipynb @@ -1,544 +1,579 @@ { - "cells": [ - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "\n", - "\n", - "\n", - "# LoRA Model Customization Job\n", - "\n", - "Learn how to use the NeMo Platform to create a LoRA (Low-Rank Adaptation) customization job using a custom dataset. In this tutorial we use LoRA to fine-tune a **question-answering model** from the SQuAD dataset.\n", - "\n", - "LoRA is a parameter-efficient fine-tuning method that requires fewer computational resources than full fine-tuning. If you need full model fine-tuning instead, see the [Full SFT Customization Job](./sft-customization-job) tutorial.\n", - "\n", - "**Time to complete:** approximately 45 minutes. Job duration increases with model size and dataset size.\n" - ] + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "\n", + "\n", + "# LoRA Model Customization Job\n", + "\n", + "Learn how to use the NeMo Platform to create a LoRA (Low-Rank Adaptation) customization job using a custom dataset. In this tutorial we use LoRA to fine-tune a **question-answering model** from the SQuAD dataset.\n", + "\n", + "LoRA is a parameter-efficient fine-tuning method that requires fewer computational resources than full fine-tuning. If you need full model fine-tuning instead, see the [Full SFT Customization Job](./sft-customization-job) tutorial.\n", + "\n", + "**Time to complete:** approximately 45 minutes. Job duration increases with model size and dataset size.\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Prerequisites\n", + "\n", + "Before starting this tutorial, ensure you have:\n", + "\n", + "1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n", + "2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n", + "3. **Installed the `datasets` package** for loading SQuAD: `pip install datasets`\n", + "4. **At least one GPU with CUDA 12.8+**\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Quick Start\n", + "\n", + "### 1. Initialize SDK\n", + "\n", + "The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used. If NeMo Platform is running elsewhere, set the `NMP_BASE_URL` environment variable:\n", + "\n", + "```sh\n", + "export NMP_BASE_URL=\n", + "```\n" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "import json\n", + "import os\n", + "import re\n", + "import time\n", + "import uuid\n", + "from pathlib import Path\n", + "from nemo_platform import NeMoPlatform, ConflictError\n", + "from nemo_platform.types.secrets import PlatformSecretResponse\n", + "from nemo_platform.types.files import HuggingfaceStorageConfigParam\n", + "\n", + "\n", + "def sanitize_name(prefix: str, name: str):\n", + " \"\"\"Sanitize model_name for deployment/config naming. Compatible with platform naming rules.\"\"\"\n", + " name = name.split(\"/\")[-1]\n", + " sanitized = re.sub(r\"[^a-z0-9@.+_-]\", \"-\", name.lower())\n", + " sanitized = re.sub(r\"-+\", \"-\", sanitized).strip(\"-\")\n", + " return f\"{prefix}-{sanitized}\"[:59].rstrip(\"-\")\n", + "\n", + "\n", + "def max_wait_time_checker(seconds: int, job_name: str = \"\"):\n", + " \"\"\"Return a check() that raises TimeoutError if called after `seconds` have elapsed.\"\"\"\n", + " start_time = time.time()\n", + "\n", + " def check():\n", + " if time.time() - start_time > seconds:\n", + " raise TimeoutError(f\"{job_name} took longer than {seconds} seconds\")\n", + "\n", + " return check\n", + "\n", + "\n", + "NMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\n", + "client = NeMoPlatform(\n", + " base_url=NMP_BASE_URL,\n", + " workspace=\"default\"\n", + ")\n" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2. Prepare Dataset\n", + "\n", + "Create your data in JSONL format (one JSON object per line). For SFT with LoRA, the platform expects **prompt/completion** pairs.\n", + "\n", + "**Dataset structure:**\n", + "- Training files under `training/` (or root with `training.jsonl`)\n", + "- Validation files under `validation/` (or `validation.jsonl`)\n", + "\n", + "**SFT format:** Each line is a JSON object with:\n", + "- **`prompt`**: The input (e.g. context + question)\n", + "- **`completion`**: The desired model output\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Example record (single line in `.jsonl`):\n", + "\n", + "```json\n", + "{\"prompt\": \"Context: ... Question: What is X? Answer:\", \"completion\": \"X is ...\"}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### Download SQuAD and Convert to SFT Format\n", + "\n", + "We use the [SQuAD](https://huggingface.co/datasets/rajpurkar/squad) dataset and convert it to prompt/completion JSONL.\n" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "from datasets import load_dataset, DatasetDict\n", + "\n", + "print(\"Loading dataset rajpurkar/squad\")\n", + "raw_dataset = load_dataset(\"rajpurkar/squad\")\n", + "if not isinstance(raw_dataset, DatasetDict):\n", + " raise ValueError(\"Dataset does not contain expected splits\")\n", + "print(\"Loaded dataset\")\n", + "\n", + "VALIDATION_PROPORTION = 0.05\n", + "SEED = 1234\n", + "training_size = 3000\n", + "validation_size = 300\n", + "DATASET_NAME = \"sft-dataset\"\n", + "DATASET_PATH = Path(\"sft-dataset\").absolute()\n", + "\n", + "os.makedirs(DATASET_PATH, exist_ok=True)\n", + "train_set = raw_dataset.get(\"train\")\n", + "split_dataset = train_set.train_test_split(test_size=VALIDATION_PROPORTION, seed=SEED)\n", + "train_ds = split_dataset[\"train\"].select(range(min(training_size, len(split_dataset[\"train\"]))))\n", + "validation_ds = split_dataset[\"test\"].select(range(min(validation_size, len(split_dataset[\"test\"]))))\n", + "\n", + "def convert_squad_to_sft_format(example):\n", + " prompt = f\"Context: {example['context']} Question: {example['question']} Answer:\"\n", + " completion = example[\"answers\"][\"text\"][0]\n", + " return {\"prompt\": prompt, \"completion\": completion}\n", + "\n", + "with open(f\"{DATASET_PATH}/training.jsonl\", \"w\", encoding=\"utf-8\") as f:\n", + " for example in train_ds:\n", + " f.write(json.dumps(convert_squad_to_sft_format(example)) + \"\\n\")\n", + "with open(f\"{DATASET_PATH}/validation.jsonl\", \"w\", encoding=\"utf-8\") as f:\n", + " for example in validation_ds:\n", + " f.write(json.dumps(convert_squad_to_sft_format(example)) + \"\\n\")\n", + "\n", + "print(f\"Saved training.jsonl with {len(train_ds)} rows\")\n", + "print(f\"Saved validation.jsonl with {len(validation_ds)} rows\")\n", + "with open(f\"{DATASET_PATH}/training.jsonl\", \"r\") as f:\n", + " sample = json.loads(f.readline())\n", + "print(\"Sample prompt (first 200 chars):\", sample[\"prompt\"][:200] + \"...\")\n", + "print(\"Sample completion:\", sample[\"completion\"])\n" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 3. Create FileSet and Upload Training Data\n", + "\n", + "Upload the training and validation JSONL files to a FileSet so the customization job can use them.\n" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "try:\n", + " client.files.filesets.create(\n", + " workspace=\"default\",\n", + " name=DATASET_NAME,\n", + " description=\"SFT training data\",\n", + " cache=True,\n", + " )\n", + " print(f\"Created fileset: {DATASET_NAME}\")\n", + "except ConflictError:\n", + " print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n", + "\n", + "client.files.fsspec.put(\n", + " lpath=DATASET_PATH,\n", + " rpath=f\"default/{DATASET_NAME}/\",\n", + " recursive=True\n", + ")\n", + "print(\"Training data:\")\n", + "print(client.files.list(fileset=DATASET_NAME, workspace=\"default\"))\n" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 4. Secrets Setup\n", + "\n", + "For Huggingface models that require authentication, create a secret with your HF token. Get a token from [Huggingface Settings](https://huggingface.co/settings/tokens) and accept the model terms.\n", + "\n", + "This is generally true for LLaMa based models (e.g. [Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct)).\n", + "\n", + "```sh\n", + "export HF_TOKEN=\n", + "```\n" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "HF_TOKEN = os.getenv(\"HF_TOKEN\")\n", + "\n", + "def create_or_get_secret(name: str, value: str | None, label: str) -> PlatformSecretResponse | None:\n", + " if not value:\n", + " print(f\"{label} is not set - skipping setting secret\")\n", + " return None\n", + " try:\n", + " secret = client.secrets.create(name=name, workspace=\"default\", value=value)\n", + " print(f\"Created secret: {name}\")\n", + " print(secret.model_dump_json(indent=2))\n", + " return secret\n", + " except ConflictError:\n", + " print(f\"Secret '{name}' already exists, continuing...\")\n", + " secret = client.secrets.retrieve(name=name, workspace=\"default\")\n", + " print(secret.model_dump_json(indent=2))\n", + " return secret\n", + "\n", + "\n", + "hf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 5. Create Base Model FileSet and Model Entity\n", + "\n", + "Create a fileset pointing to [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) and a Model Entity that references it. Model download happens when the customization job runs.\n" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "HF_REPO_ID = \"Qwen/Qwen3-0.6B\"\n", + "MODEL_NAME = \"qwen3-0.6b\"\n", + "\n", + "try:\n", + " storage = HuggingfaceStorageConfigParam(\n", + " type=\"huggingface\",\n", + " repo_id=HF_REPO_ID,\n", + " repo_type=\"model\",\n", + " )\n", + " if hf_secret:\n", + " storage[\"token_secret\"] = hf_secret.name\n", + " base_model_fs = client.files.filesets.create(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " description=\"Qwen3 0.6b base model from Huggingface\",\n", + " storage=storage,\n", + " cache=True,\n", + " )\n", + "except ConflictError:\n", + " base_model_fs = client.files.filesets.retrieve(workspace=\"default\", name=MODEL_NAME)\n", + "\n", + "try:\n", + " base_model = client.models.create(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " fileset=f\"default/{MODEL_NAME}\",\n", + " trust_remote_code=False,\n", + " )\n", + "except ConflictError:\n", + " client.models.update(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " fileset=f\"default/{MODEL_NAME}\",\n", + " trust_remote_code=False,\n", + " )\n", + " base_model = client.models.retrieve(workspace=\"default\", name=MODEL_NAME)\n", + "\n", + "print(f\"Base model fileset: fileset://default/{base_model.name}\")\n", + "print(client.files.list(fileset=MODEL_NAME, workspace=\"default\"))\n", + "\n", + "time_check = max_wait_time_checker(600, \"Model Spec\")\n", + "while not base_model.spec:\n", + " time_check()\n", + " time.sleep(10)\n", + " base_model = client.models.retrieve(workspace=\"default\", name=MODEL_NAME)\n", + "\n", + "# Clear verbose linear_layers list for cleaner output\n", + "base_model.spec.linear_layers = None\n", + "print(f\"ModelSpec: {base_model.spec}\")\n" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 6. Create LoRA Customization Job\n", + "\n", + "Submit to the **Automodel** backend using `AutomodelJobInput` with `finetuning_type: lora`. After training completes, deploy the base model with LoRA support manually (step 8).\n", + "\n" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "from nemo_automodel_plugin.schema import AutomodelJobInput\n", + "\n", + "job_suffix = uuid.uuid4().hex[:4]\n", + "JOB_NAME = f\"my-sft-job-{job_suffix}\"\n", + "OUTPUT_NAME = f\"lora-adapter-{job_suffix}\"\n", + "\n", + "spec = AutomodelJobInput(\n", + " model=f\"default/{base_model.name}\",\n", + " dataset={\"training\": f\"default/{DATASET_NAME}\"},\n", + " training={\n", + " \"training_type\": \"sft\",\n", + " \"finetuning_type\": \"lora\",\n", + " \"max_seq_length\": 2048,\n", + " },\n", + " schedule={\"epochs\": 2},\n", + " batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n", + " optimizer={\"learning_rate\": 5e-5},\n", + " parallelism={\n", + " \"num_gpus_per_node\": 1,\n", + " \"num_nodes\": 1,\n", + " \"tensor_parallel_size\": 1,\n", + " \"pipeline_parallel_size\": 1,\n", + " \"context_parallel_size\": 1,\n", + " \"expert_parallel_size\": 1,\n", + " },\n", + " output={\"name\": OUTPUT_NAME},\n", + ")\n", + "\n", + "job = client.customization.automodel.jobs.create(\n", + " spec=spec, workspace=\"default\", name=JOB_NAME\n", + ")\n", + "print(f\"Submitted job: {job.job.name}\")\n", + "print(f\"Output adapter: {OUTPUT_NAME}\")\n" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 7. Track Training Progress\n", + "\n", + "Poll job status until it completes. Progress (step/max_steps) is shown when available.\n" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "from IPython.display import clear_output\n", + "\n", + "time_check = max_wait_time_checker(3600, \"Customization Job\")\n", + "while True:\n", + " time_check()\n", + " status = client.jobs.get_status(name=job.job.name, workspace=\"default\")\n", + " clear_output(wait=True)\n", + " print(f\"Job Status: {status.status}\")\n", + " step = max_steps = training_phase = None\n", + " for job_step in status.steps or []:\n", + " if job_step.name == \"training\":\n", + " for task in job_step.tasks or []:\n", + " d = task.status_details or {}\n", + " step, max_steps = d.get(\"step\"), d.get(\"max_steps\")\n", + " training_phase = d.get(\"phase\")\n", + " break\n", + " break\n", + " if step is not None and max_steps is not None:\n", + " print(f\"Training: Step {step}/{max_steps} ({100 * step / max_steps:.1f}%)\")\n", + " if training_phase:\n", + " print(f\"Phase: {training_phase}\")\n", + " if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n", + " print(f\"\\nJob finished: {status.status}\")\n", + " break\n", + " time.sleep(10)\n", + "\n", + "assert status.status == \"completed\"\n" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 8. Validate Output Model and Deployment\n", + "\n", + "With the base model entity and LoRA adapter from training, create a NIM deployment with `lora_enabled=True` so the adapter is served alongside the base weights. Check the model entity and deployment status.\n" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "deploy_suffix = uuid.uuid4().hex[:4]\n", + "DEPLOYMENT_CONFIG_NAME = f\"lora-deploy-cfg-{deploy_suffix}\"\n", + "deployment_name = f\"lora-deploy-{deploy_suffix}\"\n", + "\n", + "deployment_config = client.inference.deployment_configs.create(\n", + " workspace=\"default\",\n", + " name=DEPLOYMENT_CONFIG_NAME,\n", + " engine=\"vllm\",\n", + " model_spec={\n", + " \"model_namespace\": \"default\",\n", + " \"model_name\": MODEL_NAME,\n", + " \"lora_enabled\": True,\n", + " },\n", + " executor_config={\n", + " \"gpu\": 1,\n", + " \"image_name\": \"vllm/vllm-openai\",\n", + " \"image_tag\": \"v0.22.1\",\n", + " \"additional_args\": [\"--max-lora-rank\", \"32\"],\n", + " },\n", + ")\n", + "\n", + "deployment = client.inference.deployments.create(\n", + " workspace=\"default\",\n", + " name=deployment_name,\n", + " config=deployment_config.name,\n", + ")\n", + "\n", + "model_entity = client.models.retrieve(workspace=\"default\", name=MODEL_NAME)\n", + "if model_entity.spec:\n", + " model_entity.spec.linear_layers = None\n", + "print(model_entity.model_dump_json(indent=2))\n", + "print(f\"Deployment status: {deployment.status}\")\n" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 9. Monitor Deployment Until Ready\n", + "\n", + "Wait for the deployment to reach RUNNING/READY before sending inference requests.\n" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "TIMEOUT_MINUTES = 30\n", + "start_time = time.time()\n", + "time_check = max_wait_time_checker(TIMEOUT_MINUTES * 60, \"Deployment\")\n", + "print(f\"Monitoring deployment '{deployment_name}'... (timeout {TIMEOUT_MINUTES} min)\\n\")\n", + "\n", + "while True:\n", + " time.sleep(15)\n", + " time_check()\n", + " deployment_status = client.inference.deployments.retrieve(name=deployment_name, workspace=\"default\")\n", + " elapsed = time.time() - start_time\n", + " clear_output(wait=True)\n", + " print(f\"Deployment: {deployment_name}\")\n", + " print(f\"Status: {deployment_status.status}\")\n", + " print(f\"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s\")\n", + " if deployment_status.status in (\"RUNNING\", \"READY\"):\n", + " print(\"\\nDeployment is ready!\")\n", + " if not client.models.wait_for_gateway(deployment_name, workspace=\"default\", timeout=60):\n", + " raise RuntimeError(\"Inference gateway did not become ready\")\n", + " break\n", + " if deployment_status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\"):\n", + " raise RuntimeError(f\"Deployment failed with status: {deployment_status.status}\")\n", + "\n", + "assert deployment_status.status in (\"RUNNING\", \"READY\")\n" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 10. Check Model Output\n", + "\n", + "Send a chat completion request to the deployed LoRA model and compare the output to the expected answer.\n" + ] + }, + { + "cell_type": "code", + "metadata": {}, + "source": [ + "context = \"The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit.\"\n", + "question = \"Who was the first person to walk on the Moon?\"\n", + "messages = [\n", + " {\"role\": \"user\", \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\"}\n", + "]\n", + "response = client.inference.gateway.provider.post(\n", + " \"v1/chat/completions\",\n", + " name=deployment_name,\n", + " workspace=\"default\",\n", + " body={\n", + " \"model\": OUTPUT_NAME,\n", + " \"messages\": messages,\n", + " \"temperature\": 0,\n", + " \"max_tokens\": 256,\n", + " }\n", + ")\n", + "print(\"=\" * 60)\n", + "print(\"MODEL INFERENCE\")\n", + "print(\"=\" * 60)\n", + "print(f\"Question: {question}\")\n", + "print(f\"Expected: Neil Armstrong\")\n", + "print(f\"Model output: {response['choices'][0]['message']['content']}\")\n" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Conclusion\n", + "\n", + "You have started a LoRA customization job, monitored it to completion, and evaluated the fine-tuned model. Use the `output.name` to access the model for further inference or evaluation.\n", + "\n", + "## Next Steps\n", + "\n", + "- [Monitor training metrics](fine-tune-metrics) in detail\n", + "- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n", + "- Try [Full SFT](./sft-customization-job) for other customization options\n", + "" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": ".venv", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.11.14" + } }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Prerequisites\n", - "\n", - "Before starting this tutorial, ensure you have:\n", - "\n", - "1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n", - "2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n", - "3. **Installed the `datasets` package** for loading SQuAD: `pip install datasets`\n", - "4. **At least one GPU with CUDA 12.8+**\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Quick Start\n", - "\n", - "### 1. Initialize SDK\n", - "\n", - "The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used. If NeMo Platform is running elsewhere, set the `NMP_BASE_URL` environment variable:\n", - "\n", - "```sh\n", - "export NMP_BASE_URL=\n", - "```\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import os\n", - "import json\n", - "import re\n", - "import time\n", - "import uuid\n", - "from pathlib import Path\n", - "from nemo_platform import NeMoPlatform, ConflictError\n", - "from nemo_platform.types.secrets import PlatformSecretResponse\n", - "from nemo_platform.types.files import HuggingfaceStorageConfigParam\n", - "from nemo_platform.types.customization import (\n", - " CustomizationJobInputParam,\n", - " DeploymentParamsParam,\n", - " LoRaParamsParam,\n", - " ParallelismParamsParam,\n", - " SftTrainingParam,\n", - ")\n", - "\n", - "\n", - "def sanitize_name(prefix: str, name: str):\n", - " \"\"\"Sanitize model_name for deployment/config naming. Compatible with platform naming rules.\"\"\"\n", - " name = name.split(\"/\")[-1]\n", - " sanitized = re.sub(r\"[^a-z0-9@.+_-]\", \"-\", name.lower())\n", - " sanitized = re.sub(r\"-+\", \"-\", sanitized).strip(\"-\")\n", - " return f\"{prefix}-{sanitized}\"[:59].rstrip(\"-\")\n", - "\n", - "\n", - "def max_wait_time_checker(seconds: int, job_name: str = \"\"):\n", - " \"\"\"Return a check() that raises TimeoutError if called after `seconds` have elapsed.\"\"\"\n", - " start_time = time.time()\n", - "\n", - " def check():\n", - " if time.time() - start_time > seconds:\n", - " raise TimeoutError(f\"{job_name} took longer than {seconds} seconds\")\n", - "\n", - " return check\n", - "\n", - "\n", - "NMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\n", - "client = NeMoPlatform(\n", - " base_url=NMP_BASE_URL,\n", - " workspace=\"default\"\n", - ")\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 2. Prepare Dataset\n", - "\n", - "Create your data in JSONL format (one JSON object per line). For SFT with LoRA, the platform expects **prompt/completion** pairs.\n", - "\n", - "**Dataset structure:**\n", - "- Training files under `training/` (or root with `training.jsonl`)\n", - "- Validation files under `validation/` (or `validation.jsonl`)\n", - "\n", - "**SFT format:** Each line is a JSON object with:\n", - "- **`prompt`**: The input (e.g. context + question)\n", - "- **`completion`**: The desired model output\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "Example record (single line in `.jsonl`):\n", - "\n", - "```json\n", - "{\"prompt\": \"Context: ... Question: What is X? Answer:\", \"completion\": \"X is ...\"}\n", - "```\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### Download SQuAD and Convert to SFT Format\n", - "\n", - "We use the [SQuAD](https://huggingface.co/datasets/rajpurkar/squad) dataset and convert it to prompt/completion JSONL.\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from datasets import load_dataset, DatasetDict\n", - "\n", - "print(\"Loading dataset rajpurkar/squad\")\n", - "raw_dataset = load_dataset(\"rajpurkar/squad\")\n", - "if not isinstance(raw_dataset, DatasetDict):\n", - " raise ValueError(\"Dataset does not contain expected splits\")\n", - "print(\"Loaded dataset\")\n", - "\n", - "VALIDATION_PROPORTION = 0.05\n", - "SEED = 1234\n", - "training_size = 3000\n", - "validation_size = 300\n", - "DATASET_NAME = \"sft-dataset\"\n", - "DATASET_PATH = Path(\"sft-dataset\").absolute()\n", - "\n", - "os.makedirs(DATASET_PATH, exist_ok=True)\n", - "train_set = raw_dataset.get(\"train\")\n", - "split_dataset = train_set.train_test_split(test_size=VALIDATION_PROPORTION, seed=SEED)\n", - "train_ds = split_dataset[\"train\"].select(range(min(training_size, len(split_dataset[\"train\"]))))\n", - "validation_ds = split_dataset[\"test\"].select(range(min(validation_size, len(split_dataset[\"test\"]))))\n", - "\n", - "def convert_squad_to_sft_format(example):\n", - " prompt = f\"Context: {example['context']} Question: {example['question']} Answer:\"\n", - " completion = example[\"answers\"][\"text\"][0]\n", - " return {\"prompt\": prompt, \"completion\": completion}\n", - "\n", - "with open(f\"{DATASET_PATH}/training.jsonl\", \"w\", encoding=\"utf-8\") as f:\n", - " for example in train_ds:\n", - " f.write(json.dumps(convert_squad_to_sft_format(example)) + \"\\n\")\n", - "with open(f\"{DATASET_PATH}/validation.jsonl\", \"w\", encoding=\"utf-8\") as f:\n", - " for example in validation_ds:\n", - " f.write(json.dumps(convert_squad_to_sft_format(example)) + \"\\n\")\n", - "\n", - "print(f\"Saved training.jsonl with {len(train_ds)} rows\")\n", - "print(f\"Saved validation.jsonl with {len(validation_ds)} rows\")\n", - "with open(f\"{DATASET_PATH}/training.jsonl\", \"r\") as f:\n", - " sample = json.loads(f.readline())\n", - "print(\"Sample prompt (first 200 chars):\", sample[\"prompt\"][:200] + \"...\")\n", - "print(\"Sample completion:\", sample[\"completion\"])\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 3. Create FileSet and Upload Training Data\n", - "\n", - "Upload the training and validation JSONL files to a FileSet so the customization job can use them.\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "try:\n", - " client.files.filesets.create(\n", - " workspace=\"default\",\n", - " name=DATASET_NAME,\n", - " description=\"SFT training data\",\n", - " cache=True,\n", - " )\n", - " print(f\"Created fileset: {DATASET_NAME}\")\n", - "except ConflictError:\n", - " print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n", - "\n", - "client.files.fsspec.put(\n", - " lpath=DATASET_PATH,\n", - " rpath=f\"default/{DATASET_NAME}/\",\n", - " recursive=True\n", - ")\n", - "print(\"Training data:\")\n", - "print(client.files.list(fileset=DATASET_NAME, workspace=\"default\"))\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 4. Secrets Setup\n", - "\n", - "For Huggingface models that require authentication, create a secret with your HF token. Get a token from [Huggingface Settings](https://huggingface.co/settings/tokens) and accept the model terms.\n", - "\n", - "This is generally true for LLaMa based models (e.g. [Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct)).\n", - "\n", - "```sh\n", - "export HF_TOKEN=\n", - "```\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": "HF_TOKEN = os.getenv(\"HF_TOKEN\")\n\ndef create_or_get_secret(name: str, value: str | None, label: str) -> PlatformSecretResponse | None:\n if not value:\n print(f\"{label} is not set - skipping setting secret\")\n return None\n try:\n secret = client.secrets.create(name=name, workspace=\"default\", value=value)\n print(f\"Created secret: {name}\")\n print(secret.model_dump_json(indent=2))\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n secret = client.secrets.retrieve(name=name, workspace=\"default\")\n print(secret.model_dump_json(indent=2))\n return secret\n\n\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")" - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 5. Create Base Model FileSet and Model Entity\n", - "\n", - "Create a fileset pointing to [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) and a Model Entity that references it. Model download happens when the customization job runs.\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "HF_REPO_ID = \"Qwen/Qwen3-0.6B\"\n", - "MODEL_NAME = \"qwen3-0.6b\"\n", - "\n", - "try:\n", - " storage = HuggingfaceStorageConfigParam(\n", - " type=\"huggingface\",\n", - " repo_id=HF_REPO_ID,\n", - " repo_type=\"model\",\n", - " )\n", - " if hf_secret:\n", - " storage[\"token_secret\"] = hf_secret.name\n", - " base_model_fs = client.files.filesets.create(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " description=\"Qwen3 0.6b base model from Huggingface\",\n", - " storage=storage,\n", - " cache=True,\n", - " )\n", - "except ConflictError:\n", - " base_model_fs = client.files.filesets.retrieve(workspace=\"default\", name=MODEL_NAME)\n", - "\n", - "try:\n", - " base_model = client.models.create(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " fileset=f\"default/{MODEL_NAME}\",\n", - " trust_remote_code=False,\n", - " )\n", - "except ConflictError:\n", - " client.models.update(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " fileset=f\"default/{MODEL_NAME}\",\n", - " trust_remote_code=False,\n", - " )\n", - " base_model = client.models.retrieve(workspace=\"default\", name=MODEL_NAME)\n", - "\n", - "print(f\"Base model fileset: fileset://default/{base_model.name}\")\n", - "print(client.files.list(fileset=MODEL_NAME, workspace=\"default\"))\n", - "\n", - "time_check = max_wait_time_checker(600, \"Model Spec\")\n", - "while not base_model.spec:\n", - " time_check()\n", - " time.sleep(10)\n", - " base_model = client.models.retrieve(workspace=\"default\", name=MODEL_NAME)\n", - "\n", - "# Clear verbose linear_layers list for cleaner output\n", - "base_model.spec.linear_layers = None\n", - "print(f\"ModelSpec: {base_model.spec}\")\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 6. Create LoRA Customization Job\n", - "\n", - "Submit a customization job with `training=SftTrainingParam(type=\"sft\", peft=LoRaParamsParam(type=\"lora\"), ...)`. Set `lora_enabled=True` in the `deployment_config` so the platform can deploy the base model with LoRA support automatically.\n", - "\n", - "When LoRA Enabled is set to true for Models Deployed via the `/apis/models` endpoint or via the `deployment_config` option during the customization job, all LoRA adapters (enabled by default) will get automatically deployed in the NIM." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "job_suffix = uuid.uuid4().hex[:4]\n", - "JOB_NAME = f\"my-sft-job-{job_suffix}\"\n", - "\n", - "job = client.customization.jobs.create(\n", - " name=JOB_NAME,\n", - " workspace=\"default\",\n", - " spec=CustomizationJobInputParam(\n", - " model=f\"default/{base_model.name}\",\n", - " dataset=f\"fileset://default/{DATASET_NAME}\",\n", - " training=SftTrainingParam(\n", - " type=\"sft\",\n", - " epochs=2,\n", - " batch_size=64,\n", - " learning_rate=0.00005,\n", - " max_seq_length=2048,\n", - " parallelism=ParallelismParamsParam(\n", - " num_gpus_per_node=1,\n", - " num_nodes=1,\n", - " tensor_parallel_size=1,\n", - " pipeline_parallel_size=1,\n", - " context_parallel_size=1,\n", - " expert_parallel_size=1,\n", - " ),\n", - " micro_batch_size=1,\n", - " peft=LoRaParamsParam(type=\"lora\"),\n", - " ),\n", - " deployment_config=DeploymentParamsParam(\n", - " lora_enabled=True,\n", - " gpu=1,\n", - " additional_envs={\"NIM_MODEL_PROFILE\": \"vllm-lora\"},\n", - " ),\n", - " ),\n", - ")\n", - "print(f\"Job ID: {job.name}\")\n", - "print(f\"Output model: {job.spec.output.name}\")\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 7. Track Training Progress\n", - "\n", - "Poll job status until it completes. Progress (step/max_steps) is shown when available.\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from IPython.display import clear_output\n", - "\n", - "time_check = max_wait_time_checker(3600, \"Customization Job\")\n", - "while True:\n", - " time_check()\n", - " status = client.customization.jobs.get_status(name=job.name, workspace=\"default\")\n", - " clear_output(wait=True)\n", - " print(f\"Job Status: {status.status}\")\n", - " step = max_steps = training_phase = None\n", - " for job_step in status.steps or []:\n", - " if job_step.name == \"customization-training-job\":\n", - " for task in job_step.tasks or []:\n", - " d = task.status_details or {}\n", - " step, max_steps = d.get(\"step\"), d.get(\"max_steps\")\n", - " training_phase = d.get(\"phase\")\n", - " break\n", - " break\n", - " if step is not None and max_steps is not None:\n", - " print(f\"Training: Step {step}/{max_steps} ({100 * step / max_steps:.1f}%)\")\n", - " if training_phase:\n", - " print(f\"Phase: {training_phase}\")\n", - " if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n", - " print(f\"\\nJob finished: {status.status}\")\n", - " break\n", - " time.sleep(10)\n", - "\n", - "assert status.status == \"completed\"\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 8. Validate Output Model and Deployment\n", - "\n", - "With `deployment_config` configured, the platform will create a NIM deployment for the base model after training. The fine-tuned LoRA adapter is enabled by default and automatically served through the deployment. Check the model entity and deployment status.\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "model_entity = client.models.retrieve(workspace=\"default\", name=MODEL_NAME)\n", - "# Clear verbose linear_layers list for cleaner output\n", - "model_entity.spec.linear_layers = None\n", - "print(model_entity.model_dump_json(indent=2))\n", - "\n", - "deployment_name = sanitize_name(\"sft-deploy\", job.spec.model)\n", - "deployment_status = client.inference.deployments.retrieve(name=deployment_name, workspace=\"default\")\n", - "print(f\"Deployment status: {deployment_status.status}\")\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 9. Monitor Deployment Until Ready\n", - "\n", - "Wait for the deployment to reach RUNNING/READY before sending inference requests.\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "TIMEOUT_MINUTES = 30\n", - "start_time = time.time()\n", - "time_check = max_wait_time_checker(TIMEOUT_MINUTES * 60, \"Deployment\")\n", - "print(f\"Monitoring deployment '{deployment_name}'... (timeout {TIMEOUT_MINUTES} min)\\n\")\n", - "\n", - "while True:\n", - " time.sleep(15)\n", - " time_check()\n", - " deployment_status = client.inference.deployments.retrieve(name=deployment_name, workspace=\"default\")\n", - " elapsed = time.time() - start_time\n", - " clear_output(wait=True)\n", - " print(f\"Deployment: {deployment_name}\")\n", - " print(f\"Status: {deployment_status.status}\")\n", - " print(f\"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s\")\n", - " if deployment_status.status in (\"RUNNING\", \"READY\"):\n", - " print(\"\\nDeployment is ready!\")\n", - " if not client.models.wait_for_gateway(deployment_name, workspace=\"default\", timeout=60):\n", - " raise RuntimeError(\"Inference gateway did not become ready\")\n", - " break\n", - " if deployment_status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\"):\n", - " raise RuntimeError(f\"Deployment failed with status: {deployment_status.status}\")\n", - "\n", - "assert deployment_status.status in (\"RUNNING\", \"READY\")\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 10. Check Model Output\n", - "\n", - "Send a chat completion request to the deployed LoRA model and compare the output to the expected answer.\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "context = \"The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit.\"\n", - "question = \"Who was the first person to walk on the Moon?\"\n", - "messages = [\n", - " {\"role\": \"user\", \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\"}\n", - "]\n", - "response = client.inference.gateway.provider.post(\n", - " \"v1/chat/completions\",\n", - " name=deployment_name,\n", - " workspace=\"default\",\n", - " body={\n", - " \"model\": job.spec.output.name,\n", - " \"messages\": messages,\n", - " \"temperature\": 0,\n", - " \"max_tokens\": 256,\n", - " }\n", - ")\n", - "print(\"=\" * 60)\n", - "print(\"MODEL INFERENCE\")\n", - "print(\"=\" * 60)\n", - "print(f\"Question: {question}\")\n", - "print(f\"Expected: Neil Armstrong\")\n", - "print(f\"Model output: {response['choices'][0]['message']['content']}\")\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Conclusion\n", - "\n", - "You have started a LoRA customization job, monitored it to completion, and evaluated the fine-tuned model. Use the `output.name` to access the model for further inference or evaluation.\n", - "\n", - "## Next Steps\n", - "\n", - "- [Monitor training metrics](fine-tune-metrics) in detail\n", - "- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n", - "- Try [Full SFT](./sft-customization-job) or [DPO](./dpo-customization-job) for other customization options\n" - ] - } - ], - "metadata": { - "kernelspec": { - "display_name": ".venv", - "language": "python", - "name": "python3" - }, - "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 3 - }, - "file_extension": ".py", - "mimetype": "text/x-python", - "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython3", - "version": "3.11.14" - } - }, - "nbformat": 4, - "nbformat_minor": 4 -} + "nbformat": 4, + "nbformat_minor": 4 +} \ No newline at end of file diff --git a/docs/customizer/tutorials/lora-customization-job.mdx b/docs/customizer/tutorials/lora-customization-job.mdx index d4b3e1eaa0..d7434b20a7 100644 --- a/docs/customizer/tutorials/lora-customization-job.mdx +++ b/docs/customizer/tutorials/lora-customization-job.mdx @@ -3,7 +3,437 @@ title: "LoRA Model Customization" description: "" --- - +[Run in Google Colab](https://colab.research.google.com/github/NVIDIA-NeMo/nemo-platform/blob/main/docs/customizer/tutorials/lora-customization-job.ipynb) + +# LoRA Model Customization Job + +Learn how to use the NeMo Platform to create a LoRA (Low-Rank Adaptation) customization job using a custom dataset. In this tutorial we use LoRA to fine-tune a **question-answering model** from the SQuAD dataset. + +LoRA is a parameter-efficient fine-tuning method that requires fewer computational resources than full fine-tuning. If you need full model fine-tuning instead, see the [Full SFT Customization Job](/documentation/customizer-reference/tutorials/sft-customization-job) tutorial. + +**Time to complete:** approximately 45 minutes. Job duration increases with model size and dataset size. + +## Prerequisites + +Before starting this tutorial, ensure you have: + +1. **Completed the [Quickstart](/documentation/get-started)** to install and deploy NeMo Platform locally +2. **Installed the Python SDK** (PyPI wrapper: `pip install "nemo-platform[all]"`; source checkout: run `make bootstrap` from the repository root) +3. **Installed the `datasets` package** for loading SQuAD: `pip install datasets` +4. **At least one GPU with CUDA 12.8+** + +## Quick Start + +### 1. Initialize SDK + +The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used. If NeMo Platform is running elsewhere, set the `NMP_BASE_URL` environment variable: + +```sh +export NMP_BASE_URL= +``` + +```python +import json +import os +import re +import time +import uuid +from pathlib import Path +from nemo_platform import NeMoPlatform, ConflictError +from nemo_platform.types.secrets import PlatformSecretResponse +from nemo_platform.types.files import HuggingfaceStorageConfigParam + + +def sanitize_name(prefix: str, name: str): + """Sanitize model_name for deployment/config naming. Compatible with platform naming rules.""" + name = name.split("/")[-1] + sanitized = re.sub(r"[^a-z0-9@.+_-]", "-", name.lower()) + sanitized = re.sub(r"-+", "-", sanitized).strip("-") + return f"{prefix}-{sanitized}"[:59].rstrip("-") + + +def max_wait_time_checker(seconds: int, job_name: str = ""): + """Return a check() that raises TimeoutError if called after `seconds` have elapsed.""" + start_time = time.time() + + def check(): + if time.time() - start_time > seconds: + raise TimeoutError(f"{job_name} took longer than {seconds} seconds") + + return check + + +NMP_BASE_URL = os.environ.get("NMP_BASE_URL", "http://localhost:8080") +client = NeMoPlatform( + base_url=NMP_BASE_URL, + workspace="default" +) + +``` + +### 2. Prepare Dataset + +Create your data in JSONL format (one JSON object per line). For SFT with LoRA, the platform expects **prompt/completion** pairs. + +**Dataset structure:** +- Training files under `training/` (or root with `training.jsonl`) +- Validation files under `validation/` (or `validation.jsonl`) + +**SFT format:** Each line is a JSON object with: +- **`prompt`**: The input (e.g. context + question) +- **`completion`**: The desired model output + +Example record (single line in `.jsonl`): + +```json +{"prompt": "Context: ... Question: What is X? Answer:", "completion": "X is ..."} +``` + +#### Download SQuAD and Convert to SFT Format + +We use the [SQuAD](https://huggingface.co/datasets/rajpurkar/squad) dataset and convert it to prompt/completion JSONL. + +```python +from datasets import load_dataset, DatasetDict + +print("Loading dataset rajpurkar/squad") +raw_dataset = load_dataset("rajpurkar/squad") +if not isinstance(raw_dataset, DatasetDict): + raise ValueError("Dataset does not contain expected splits") +print("Loaded dataset") + +VALIDATION_PROPORTION = 0.05 +SEED = 1234 +training_size = 3000 +validation_size = 300 +DATASET_NAME = "sft-dataset" +DATASET_PATH = Path("sft-dataset").absolute() + +os.makedirs(DATASET_PATH, exist_ok=True) +train_set = raw_dataset.get("train") +split_dataset = train_set.train_test_split(test_size=VALIDATION_PROPORTION, seed=SEED) +train_ds = split_dataset["train"].select(range(min(training_size, len(split_dataset["train"])))) +validation_ds = split_dataset["test"].select(range(min(validation_size, len(split_dataset["test"])))) + +def convert_squad_to_sft_format(example): + prompt = f"Context: {example['context']} Question: {example['question']} Answer:" + completion = example["answers"]["text"][0] + return {"prompt": prompt, "completion": completion} + +with open(f"{DATASET_PATH}/training.jsonl", "w", encoding="utf-8") as f: + for example in train_ds: + f.write(json.dumps(convert_squad_to_sft_format(example)) + "\n") +with open(f"{DATASET_PATH}/validation.jsonl", "w", encoding="utf-8") as f: + for example in validation_ds: + f.write(json.dumps(convert_squad_to_sft_format(example)) + "\n") + +print(f"Saved training.jsonl with {len(train_ds)} rows") +print(f"Saved validation.jsonl with {len(validation_ds)} rows") +with open(f"{DATASET_PATH}/training.jsonl", "r") as f: + sample = json.loads(f.readline()) +print("Sample prompt (first 200 chars):", sample["prompt"][:200] + "...") +print("Sample completion:", sample["completion"]) + +``` + +### 3. Create FileSet and Upload Training Data + +Upload the training and validation JSONL files to a FileSet so the customization job can use them. + +```python +try: + client.files.filesets.create( + workspace="default", + name=DATASET_NAME, + description="SFT training data", + cache=True, + ) + print(f"Created fileset: {DATASET_NAME}") +except ConflictError: + print(f"Fileset '{DATASET_NAME}' already exists, continuing...") + +client.files.fsspec.put( + lpath=DATASET_PATH, + rpath=f"default/{DATASET_NAME}/", + recursive=True +) +print("Training data:") +print(client.files.list(fileset=DATASET_NAME, workspace="default")) + +``` + +### 4. Secrets Setup + +For Huggingface models that require authentication, create a secret with your HF token. Get a token from [Huggingface Settings](https://huggingface.co/settings/tokens) and accept the model terms. + +This is generally true for LLaMa based models (e.g. [Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct)). + +```sh +export HF_TOKEN= +``` + +```python +HF_TOKEN = os.getenv("HF_TOKEN") + +def create_or_get_secret(name: str, value: str | None, label: str) -> PlatformSecretResponse | None: + if not value: + print(f"{label} is not set - skipping setting secret") + return None + try: + secret = client.secrets.create(name=name, workspace="default", value=value) + print(f"Created secret: {name}") + print(secret.model_dump_json(indent=2)) + return secret + except ConflictError: + print(f"Secret '{name}' already exists, continuing...") + secret = client.secrets.retrieve(name=name, workspace="default") + print(secret.model_dump_json(indent=2)) + return secret + + +hf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN") +``` + +### 5. Create Base Model FileSet and Model Entity + +Create a fileset pointing to [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) and a Model Entity that references it. Model download happens when the customization job runs. + +```python +HF_REPO_ID = "Qwen/Qwen3-0.6B" +MODEL_NAME = "qwen3-0.6b" + +try: + storage = HuggingfaceStorageConfigParam( + type="huggingface", + repo_id=HF_REPO_ID, + repo_type="model", + ) + if hf_secret: + storage["token_secret"] = hf_secret.name + base_model_fs = client.files.filesets.create( + workspace="default", + name=MODEL_NAME, + description="Qwen3 0.6b base model from Huggingface", + storage=storage, + cache=True, + ) +except ConflictError: + base_model_fs = client.files.filesets.retrieve(workspace="default", name=MODEL_NAME) + +try: + base_model = client.models.create( + workspace="default", + name=MODEL_NAME, + fileset=f"default/{MODEL_NAME}", + trust_remote_code=False, + ) +except ConflictError: + client.models.update( + workspace="default", + name=MODEL_NAME, + fileset=f"default/{MODEL_NAME}", + trust_remote_code=False, + ) + base_model = client.models.retrieve(workspace="default", name=MODEL_NAME) + +print(f"Base model fileset: fileset://default/{base_model.name}") +print(client.files.list(fileset=MODEL_NAME, workspace="default")) + +time_check = max_wait_time_checker(600, "Model Spec") +while not base_model.spec: + time_check() + time.sleep(10) + base_model = client.models.retrieve(workspace="default", name=MODEL_NAME) + +# Clear verbose linear_layers list for cleaner output +base_model.spec.linear_layers = None +print(f"ModelSpec: {base_model.spec}") + +``` + +### 6. Create LoRA Customization Job + +Submit to the **Automodel** backend using `AutomodelJobInput` with `finetuning_type: lora`. After training completes, deploy the base model with LoRA support manually (step 8). + +```python +from nemo_automodel_plugin.schema import AutomodelJobInput + +job_suffix = uuid.uuid4().hex[:4] +JOB_NAME = f"my-sft-job-{job_suffix}" +OUTPUT_NAME = f"lora-adapter-{job_suffix}" + +spec = AutomodelJobInput( + model=f"default/{base_model.name}", + dataset={"training": f"default/{DATASET_NAME}"}, + training={ + "training_type": "sft", + "finetuning_type": "lora", + "max_seq_length": 2048, + }, + schedule={"epochs": 2}, + batch={"global_batch_size": 64, "micro_batch_size": 1}, + optimizer={"learning_rate": 5e-5}, + parallelism={ + "num_gpus_per_node": 1, + "num_nodes": 1, + "tensor_parallel_size": 1, + "pipeline_parallel_size": 1, + "context_parallel_size": 1, + "expert_parallel_size": 1, + }, + output={"name": OUTPUT_NAME}, +) + +job = client.customization.automodel.jobs.create( + spec=spec, workspace="default", name=JOB_NAME +) +print(f"Submitted job: {job.job.name}") +print(f"Output adapter: {OUTPUT_NAME}") + +``` + +### 7. Track Training Progress + +Poll job status until it completes. Progress (step/max_steps) is shown when available. + +```python +from IPython.display import clear_output + +time_check = max_wait_time_checker(3600, "Customization Job") +while True: + time_check() + status = client.jobs.get_status(name=job.job.name, workspace="default") + clear_output(wait=True) + print(f"Job Status: {status.status}") + step = max_steps = training_phase = None + for job_step in status.steps or []: + if job_step.name == "training": + for task in job_step.tasks or []: + d = task.status_details or {} + step, max_steps = d.get("step"), d.get("max_steps") + training_phase = d.get("phase") + break + break + if step is not None and max_steps is not None: + print(f"Training: Step {step}/{max_steps} ({100 * step / max_steps:.1f}%)") + if training_phase: + print(f"Phase: {training_phase}") + if status.status in ("completed", "failed", "cancelled", "error"): + print(f"\nJob finished: {status.status}") + break + time.sleep(10) + +assert status.status == "completed" + +``` + +### 8. Validate Output Model and Deployment + +With the base model entity and LoRA adapter from training, create a NIM deployment with `lora_enabled=True` so the adapter is served alongside the base weights. Check the model entity and deployment status. + +```python +deploy_suffix = uuid.uuid4().hex[:4] +DEPLOYMENT_CONFIG_NAME = f"lora-deploy-cfg-{deploy_suffix}" +deployment_name = f"lora-deploy-{deploy_suffix}" + +deployment_config = client.inference.deployment_configs.create( + workspace="default", + name=DEPLOYMENT_CONFIG_NAME, + engine="vllm", + model_spec={ + "model_namespace": "default", + "model_name": MODEL_NAME, + "lora_enabled": True, + }, + executor_config={ + "gpu": 1, + "image_name": "vllm/vllm-openai", + "image_tag": "v0.22.1", + "additional_args": ["--max-lora-rank", "32"], + }, +) + +deployment = client.inference.deployments.create( + workspace="default", + name=deployment_name, + config=deployment_config.name, +) + +model_entity = client.models.retrieve(workspace="default", name=MODEL_NAME) +if model_entity.spec: + model_entity.spec.linear_layers = None +print(model_entity.model_dump_json(indent=2)) +print(f"Deployment status: {deployment.status}") + +``` + +### 9. Monitor Deployment Until Ready + +Wait for the deployment to reach RUNNING/READY before sending inference requests. + +```python +TIMEOUT_MINUTES = 30 +start_time = time.time() +time_check = max_wait_time_checker(TIMEOUT_MINUTES * 60, "Deployment") +print(f"Monitoring deployment '{deployment_name}'... (timeout {TIMEOUT_MINUTES} min)\n") + +while True: + time.sleep(15) + time_check() + deployment_status = client.inference.deployments.retrieve(name=deployment_name, workspace="default") + elapsed = time.time() - start_time + clear_output(wait=True) + print(f"Deployment: {deployment_name}") + print(f"Status: {deployment_status.status}") + print(f"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s") + if deployment_status.status in ("RUNNING", "READY"): + print("\nDeployment is ready!") + if not client.models.wait_for_gateway(deployment_name, workspace="default", timeout=60): + raise RuntimeError("Inference gateway did not become ready") + break + if deployment_status.status in ("FAILED", "ERROR", "TERMINATED"): + raise RuntimeError(f"Deployment failed with status: {deployment_status.status}") + +assert deployment_status.status in ("RUNNING", "READY") + +``` + +### 10. Check Model Output + +Send a chat completion request to the deployed LoRA model and compare the output to the expected answer. + +```python +context = "The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit." +question = "Who was the first person to walk on the Moon?" +messages = [ + {"role": "user", "content": f"Based on the following context, answer the question.\n\nContext: {context}\n\nQuestion: {question}"} +] +response = client.inference.gateway.provider.post( + "v1/chat/completions", + name=deployment_name, + workspace="default", + body={ + "model": OUTPUT_NAME, + "messages": messages, + "temperature": 0, + "max_tokens": 256, + } +) +print("=" * 60) +print("MODEL INFERENCE") +print("=" * 60) +print(f"Question: {question}") +print(f"Expected: Neil Armstrong") +print(f"Model output: {response['choices'][0]['message']['content']}") + +``` + +## Conclusion + +You have started a LoRA customization job, monitored it to completion, and evaluated the fine-tuned model. Use the `output.name` to access the model for further inference or evaluation. + +## Next Steps + +- [Monitor training metrics](/documentation/customizer-reference/tutorials/metrics) in detail +- [Evaluate your fine-tuned model](/documentation/evaluate-models) using the Evaluator service +- Try [Full SFT](/documentation/customizer-reference/tutorials/sft-customization-job) for other customization options diff --git a/docs/customizer/tutorials/metrics.mdx b/docs/customizer/tutorials/metrics.mdx index 752da66014..b857a39caa 100644 --- a/docs/customizer/tutorials/metrics.mdx +++ b/docs/customizer/tutorials/metrics.mdx @@ -36,7 +36,7 @@ Each customization job tracks two key metrics: ### Using the API -Get job status and training metrics using the Customization Service: +Get job status and training metrics through the platform Jobs service: ```python import os @@ -49,14 +49,14 @@ client = NeMoPlatform( # Get job status with metrics job_name = "my-sft-job" -status = client.customization.jobs.get_status(name=job_name, workspace="default") +status = client.jobs.get_status(name=job_name, workspace="default") print(f"Job: {status.name}") print(f"Status: {status.status}") # Check training step progress for step in status.steps or []: - if step.name == "customization-training-job": + if step.name == "training": for task in step.tasks or []: details = task.status_details or {} print(f"Training Phase: {details.get('phase')}") @@ -95,38 +95,33 @@ If your customization job was created with W&B integration enabled (see [Weights 3. View training and validation loss curves, learning rate schedules, and other metrics under the run's dashboard ```python -client = NeMoPlatform( - base_url=os.environ.get("NMP_BASE_URL", "http://localhost:8080"), - workspace="default", +from nemo_automodel_plugin.schema import AutomodelJobInput + +# Create an Automodel job with W&B integration +spec = AutomodelJobInput( + model="default/llama-3-2-1b", + dataset={"training": "default/my-dataset"}, + training={"training_type": "sft", "finetuning_type": "lora"}, + schedule={"epochs": 3}, + batch={"global_batch_size": 16, "micro_batch_size": 1}, + optimizer={"learning_rate": 1e-4}, + integrations={ + "wandb": { + "project": "my-finetuning-project", + "entity": "my-team", + "tags": ["fine-tuning", "llama"], + "api_key_secret": "my-wandb-key", + } + }, ) -# Create a customization job with W&B integration -job = client.customization.jobs.create( +job = client.customization.automodel.jobs.create( name="my-wandb-job", workspace="default", - spec={ - "model": "default/llama-3-2-1b", - "dataset": "fileset://default/my-dataset", - "training": { - "type": "sft", - "peft": {"type": "lora"}, - "epochs": 3, - "batch_size": 16, - "learning_rate": 1e-4, - }, - "integrations": { - "wandb": { - "project": "my-finetuning-project", - "entity": "my-team", - "tags": ["fine-tuning", "llama"], - "api_key_secret": "my-wandb-key", - } - }, - }, + spec=spec, ) -print(f"Created job: {job.name}") -print(f"Status: {job.status}") +print(f"Submitted job: {job.job.name}") ``` The `api_key_secret` field references a stored secret containing your `WANDB_API_KEY`. diff --git a/docs/customizer/tutorials/optimize-throughput.ipynb b/docs/customizer/tutorials/optimize-throughput.ipynb index cd13d5ee5a..ed69b9ce27 100644 --- a/docs/customizer/tutorials/optimize-throughput.ipynb +++ b/docs/customizer/tutorials/optimize-throughput.ipynb @@ -1,977 +1,974 @@ { - "cells": [ - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "\n", - "\n", - "\n", - "# Optimize for Tokens/GPU Throughput\n", - "\n", - "## About\n", - "Learn how to use the NeMo Platform Customizer to create a [LoRA](nemo-ms-about-concepts-customization) (Low-Rank Adaptation) customization job optimized for higher tokens/GPU throughput and lower runtime. \n", - "\n", - "**In this tutorial, you will:**\n", - "1. Fine-tune [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) on the SQuAD dataset using LoRA, with [sequence packing](nemo-ms-about-concepts-customization) enabled for one run and disabled for another.\n", - "2. Compare training runtime, GPU utilization, and memory allocation between the two runs.\n", - "3. Verify that validation loss remains comparable, confirming that sequence packing improves throughput without sacrificing model quality.\n", - "\n", - "> **Note:** While this tutorial demonstrates sequence packing with LoRA, the optimization is also available for all_weights (full) SFT customization jobs." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Prerequisites\n", - "\n", - "Before starting this tutorial, ensure you have:\n", - "\n", - "1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n", - "2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Quick Start\n", - "\n", - "### 1. Initialize SDK\n", - "\n", - "The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](../../get-started/quickstart.md) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable:\n", - "\n", - "```sh\n", - "export NMP_BASE_URL=\n", - "```" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import json\n", - "import os\n", - "from nemo_platform import NeMoPlatform, ConflictError\n", - "\n", - "NMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\n", - "client = NeMoPlatform(\n", - " base_url=NMP_BASE_URL,\n", - " workspace=\"default\"\n", - ")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 2. Create Dataset FileSet and Upload Training Data" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "Install additional dependencies if they are not installed in your Python environment.\n", - "\n", - "The cell below automatically detects your environment and uses:\n", - "- `uv pip install` if you're in a uv-managed virtual environment\n", - "- `pip install` otherwise\n", - "\n", - "Required packages:\n", - "- `datasets` - Download the public [rajpurkar/squad](https://huggingface.co/datasets/rajpurkar/squad) dataset\n", - "- `pandas` - Compare job results in table format\n", - "- `matplotlib` - Plot live training metrics (loss curves, GPU utilization)\n", - "- `nvidia-ml-py` - Collect GPU VRAM and compute utilization metrics during training" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "vscode": { - "languageId": "shellscript" - } - }, - "outputs": [], - "source": [ - "%%bash\n", - "if command -v uv >/dev/null 2>&1 && [ -n \"$VIRTUAL_ENV\" ]; then\n", - " uv pip install datasets pandas matplotlib nvidia-ml-py\n", - "else\n", - " pip install datasets pandas matplotlib nvidia-ml-py\n", - "fi" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### Download rajpurkar/squad Dataset\n", - "\n", - "SQuAD (Stanford Question Answering Dataset) is a reading comprehension dataset consisting of questions posed on Wikipedia articles, where the answer is a segment of text from the corresponding passage." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import json\n", - "import os\n", - "from pathlib import Path\n", - "from datasets import load_dataset, Dataset, DatasetDict\n", - "\n", - "# Configuration\n", - "SEED = 1234\n", - "DATASET_NAME = \"sft-dataset\"\n", - "\n", - "# Convert SQuAD format to prompt/completion format and save to JSONL\n", - "def convert_squad_to_sft_format(example):\n", - " \"\"\"Convert SQuAD format to prompt/completion format for SFT training.\"\"\"\n", - " prompt = f\"Context: {example['context']} Question: {example['question']} Answer:\"\n", - " completion = example[\"answers\"][\"text\"][0] # Take the first answer\n", - " return {\"prompt\": prompt, \"completion\": completion}\n", - "\n", - "# Load the SQuAD dataset from Hugging Face\n", - "print(\"Loading dataset rajpurkar/squad\")\n", - "ds = load_dataset(\"rajpurkar/squad\")\n", - "if not isinstance(ds, DatasetDict):\n", - " raise ValueError(\"Dataset does not contain expected splits\")\n", - "\n", - "print(\"Loaded dataset\")\n", - "\n", - "# For the purpose of this tutorial, we'll use a subset of the dataset\n", - "# We use a reduced dataset size (3000 training/300 validation samples) to keep tutorial runtime manageable\n", - "# while still demonstrating the performance benefits of sequence packing. The larger the dataset,\n", - "# the better the model will perform but the longer the training will take.\n", - "training_size = 3000\n", - "validation_size = 300\n", - "DATASET_PATH = Path(DATASET_NAME).absolute()\n", - "\n", - "# Get training split and verify it's a Dataset (not IterableDataset)\n", - "train_dataset = ds[\"train\"]\n", - "validation_dataset = ds[\"validation\"]\n", - "assert isinstance(train_dataset, Dataset), \"Expected Dataset type\"\n", - "assert isinstance(validation_dataset, Dataset), \"Expected Dataset type\"\n", - "\n", - "# Select subsets and save to JSONL files\n", - "training_ds = train_dataset.select(range(training_size))\n", - "validation_ds = validation_dataset.select(range(validation_size))\n", - "\n", - "# Transform to SFT format (prompt/completion)\n", - "training_ds = training_ds.map(convert_squad_to_sft_format, remove_columns=training_ds.column_names)\n", - "validation_ds = validation_ds.map(convert_squad_to_sft_format, remove_columns=validation_ds.column_names)\n", - "\n", - "# Create directory if it doesn't exist\n", - "# Note: This will create a local 'sft-dataset/' directory with training.jsonl and validation.jsonl files\n", - "os.makedirs(DATASET_PATH, exist_ok=True)\n", - "\n", - "# Save subsets to JSONL files\n", - "training_ds.to_json(f\"{DATASET_PATH}/training.jsonl\")\n", - "validation_ds.to_json(f\"{DATASET_PATH}/validation.jsonl\")\n", - "\n", - "print(f\"Saved training.jsonl with {len(training_ds)} rows\")\n", - "print(f\"Saved validation.jsonl with {len(validation_ds)} rows\")" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Create fileset to store SFT training data\n", - "\n", - "try:\n", - " client.files.filesets.create(\n", - " workspace=\"default\",\n", - " name=DATASET_NAME,\n", - " description=\"SFT training data\"\n", - " )\n", - " print(f\"Created fileset: {DATASET_NAME}\")\n", - "except ConflictError:\n", - " print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n", - "\n", - "# Upload training data files individually to ensure correct structure\n", - "client.files.upload(\n", - " local_path=DATASET_PATH, # Local directory with your JSONL files\n", - " remote_path=\"\",\n", - " fileset=DATASET_NAME,\n", - " workspace=\"default\"\n", - ")\n", - "\n", - "# Validate training data is uploaded correctly\n", - "print(\"Training data:\")\n", - "print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 3. Secrets Setup\n", - "\n", - "If you plan to use NGC or HuggingFace models, you will need to configure authentication:\n", - "\n", - "- **NGC models** (`ngc://` URIs): Requires NGC API key\n", - "- **HuggingFace models** (`hf://` URIs): Requires HF token for gated/private models\n", - "\n", - "\n", - "Configure these as secrets in your platform. Refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md) for detailed instructions.\n", - "\n", - "Get your credentials to access base models:\n", - "- [NGC API Key](https://ngc.nvidia.com/) (Setup → Generate API Key)\n", - "- [HuggingFace Token](https://huggingface.co/settings/tokens) (Create token with Read access)\n", - "\n", - "\n", - "---\n", - "\n", - "#### Quick Setup Example\n", - "\n", - "This tutorial uses the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model from HuggingFace. Ensure that you have sufficient permissions to download the model. If you cannot access the files on the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) Hugging Face page, request access.\n", - "\n", - "**HuggingFace Authentication:**\n", - "- For gated models (Llama, Gemma), you must provide a HuggingFace token via the `token_secret` parameter\n", - "- Get your token from [HuggingFace Settings](https://huggingface.co/settings/tokens) (requires Read access)\n", - "- Accept the model's terms on the HuggingFace model page before using it. Example: [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main)\n", - "- For public models, you can omit the `token_secret` parameter when creating a fileset for the model in the next step." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Export the HF_TOKEN and NGC_API_KEY environment variables if they are not already set\n", - "HF_TOKEN = os.getenv(\"HF_TOKEN\")\n", - "NGC_API_KEY = os.getenv(\"NGC_API_KEY\")\n", - "\n", - "\n", - "def create_or_get_secret(name: str, value: str | None, label: str):\n", - " if not value:\n", - " raise ValueError(f\"{label} environment variable is not set. Set it and try again.\")\n", - " try:\n", - " secret = client.secrets.create(\n", - " name=name,\n", - " workspace=\"default\",\n", - " value=value,\n", - " )\n", - " print(f\"Created secret: {name}\")\n", - " return secret\n", - " except ConflictError:\n", - " print(f\"Secret '{name}' already exists, continuing...\")\n", - " return client.secrets.retrieve(name=name, workspace=\"default\")\n", - "\n", - "\n", - "# Create HuggingFace token secret\n", - "hf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\n", - "print(\"HF_TOKEN secret:\")\n", - "print(hf_secret.model_dump_json(indent=2))\n", - "\n", - "# Create NGC API key secret\n", - "# Uncomment the line below if you have NGC API Key and want to finetune NGC models\n", - "# ngc_api_key = create_or_get_secret(\"ngc-api-key\", NGC_API_KEY, \"NGC_API_KEY\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 4. Create Base Model FileSet\n", - "\n", - "Create a fileset pointing to the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model on HuggingFace. This step creates a pointer to the model on Hugging Face and does not download it. The model is downloaded at job creation time.\n", - "\n", - "Note: for public models, you can omit the `token_secret` parameter when creating a model fileset." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import time\n", - "from nemo_platform.types.files import HuggingfaceStorageConfigParam\n", - "\n", - "HF_REPO_ID = \"meta-llama/Llama-3.2-1B-Instruct\"\n", - "MODEL_NAME = \"llama-3-2-1b-base\"\n", - "\n", - "# Ensure you have a HuggingFace token secret created\n", - "# Create a fileset pointing to the desired HuggingFace model\n", - "try:\n", - " base_model_fs = client.files.filesets.create(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " description=\"Llama 3.2 1B base model from HuggingFace\",\n", - " storage=HuggingfaceStorageConfigParam(\n", - " type=\"huggingface\",\n", - " # repo_id is the full model name from Hugging Face\n", - " repo_id=HF_REPO_ID,\n", - " repo_type=\"model\",\n", - " # we use the secret created in the previous step\n", - " token_secret=hf_secret.name\n", - " )\n", - " )\n", - " print(f\"Created base model fileset: {MODEL_NAME}\")\n", - "except ConflictError:\n", - " print(f\"Base model fileset already exists. Skipping creation.\")\n", - " base_model_fs = client.files.filesets.retrieve(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " )\n", - "\n", - "# Create the Model Entity representation.\n", - "try:\n", - " base_model = client.models.create(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " fileset=f\"default/{MODEL_NAME}\",\n", - " )\n", - " print(f\"Created Model Entity: {MODEL_NAME}\")\n", - "except ConflictError:\n", - " print(f\"Base model already exists. Updating fileset if different.\")\n", - " base_model = client.models.update(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " fileset=f\"default/{MODEL_NAME}\",\n", - " )\n", - "\n", - "print(f\"\\nBase model fileset: fileset://default/{base_model.name}\")\n", - "print(\"Base model fileset files list:\")\n", - "print(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=\"default\").data], indent=2))\n", - "\n", - "# Wait for ModelSpec to be populated from the checkpoint\n", - "print(\"\\nWaiting for ModelSpec to be populated...\")\n", - "SPEC_TIMEOUT_SECONDS = 120\n", - "spec_start = time.time()\n", - "while not base_model.spec:\n", - " if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n", - " raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n", - " time.sleep(2)\n", - " base_model = client.models.retrieve(\n", - " workspace=\"default\",\n", - " name=MODEL_NAME,\n", - " )\n", - "\n", - "print(f\"ModelSpec populated: {base_model.spec}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 5. Create LoRA Job with Sequence Packing\n", - "Create a customization job with an inline target referencing the base model and dataset filesets created in previous steps." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import uuid\n", - "from nemo_platform.types.customization import (\n", - " CustomizationJobInputParam,\n", - " SftTrainingParam,\n", - " ParallelismParamsParam,\n", - " LoRaParamsParam,\n", - ")\n", - "\n", - "# Enable sequence packing to improve throughput and GPU utilization\n", - "SEQUENCE_PACKING_ENABLED = True\n", - "\n", - "job_suffix = uuid.uuid4().hex[:4]\n", - "\n", - "JOB_NAME = f\"my-sft-job-{job_suffix}\"\n", - "job_spec = CustomizationJobInputParam(\n", - " model=f\"default/{base_model.name}\",\n", - " dataset=f\"fileset://default/{DATASET_NAME}\",\n", - " training=SftTrainingParam(\n", - " type=\"sft\",\n", - " epochs=1,\n", - " batch_size=64,\n", - " learning_rate=0.00005,\n", - " max_seq_length=4096,\n", - " val_check_interval=0.1,\n", - " micro_batch_size=1,\n", - " sequence_packing=SEQUENCE_PACKING_ENABLED,\n", - " peft=LoRaParamsParam(),\n", - " parallelism=ParallelismParamsParam(\n", - " num_gpus_per_node=1,\n", - " num_nodes=1,\n", - " tensor_parallel_size=1,\n", - " pipeline_parallel_size=1,\n", - " ),\n", - " )\n", - ")\n", - "\n", - "job_with_sequence_packing = client.customization.jobs.create(\n", - " name=JOB_NAME,\n", - " workspace=\"default\",\n", - " spec=job_spec\n", - ")\n", - "\n", - "print(f\"Job ID: {job_with_sequence_packing.name}\")\n", - "print(f\"Output model: {job_with_sequence_packing.spec.output.name}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 6. Track Finetuning Progress\n", - "\n", - "A training job contains multiple steps: \n", - "- Model and dataset downloading\n", - "- Finetuning where LoRA adapter weights are trained\n", - "- Creating a fileset entry for the finetuned model\n", - "- Finetuned weights uploading\n", - "\n", - "The elapsed time printed below reflects progress of the entire job. We compare the time taken by the finetuning step for both jobs in the last section of this tutorial." - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### Define Helper Functions" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "\n", - "# Helpers to draw GPU VRAM Utilization and Validation Loss\n", - "import matplotlib.pyplot as plt\n", - "try:\n", - " import pynvml\n", - " _PYNVML_AVAILABLE = True\n", - "except ImportError:\n", - " _PYNVML_AVAILABLE = False\n", - " print(\"Note: Install nvidia-ml-py ('pip install nvidia-ml-py' or 'uv pip install nvidia-ml-py') to enable live GPU metrics.\")\n", - "\n", - "# ---------------------------------------------------------------------------\n", - "# GPU metrics collection (nvidia-ml-py; import name is pynvml)\n", - "# ---------------------------------------------------------------------------\n", - "\n", - "def _get_gpu_snapshot() -> tuple[list[float], list[float]]:\n", - " \"\"\"Return (vram_usage_pcts, compute_util_pcts) for each GPU.\"\"\"\n", - " if not _PYNVML_AVAILABLE:\n", - " return [], []\n", - " pynvml.nvmlInit()\n", - " try:\n", - " vram, util = [], []\n", - " for i in range(pynvml.nvmlDeviceGetCount()):\n", - " h = pynvml.nvmlDeviceGetHandleByIndex(i)\n", - " mem = pynvml.nvmlDeviceGetMemoryInfo(h)\n", - " rates = pynvml.nvmlDeviceGetUtilizationRates(h)\n", - " vram.append(int(mem.used) / int(mem.total) * 100)\n", - " util.append(float(rates.gpu))\n", - " return vram, util\n", - " finally:\n", - " pynvml.nvmlShutdown()\n", - "\n", - "\n", - "# ---------------------------------------------------------------------------\n", - "# Dashboard drawing helpers\n", - "# ---------------------------------------------------------------------------\n", - "\n", - "_PALETTE = {\n", - " \"val_loss\": \"#E74C3C\",\n", - " \"train_loss\": \"#F39C12\",\n", - " \"vram\": [\"#3498DB\", \"#9B59B6\", \"#1ABC9C\", \"#E67E22\"],\n", - " \"util\": [\"#2ECC71\", \"#E74C3C\", \"#3498DB\", \"#F1C40F\"],\n", - " \"grid\": \"#ECECEC\",\n", - " \"title\": \"#2C3E50\",\n", - " \"subtitle\": \"#7F8C8D\",\n", - " \"spine\": \"#CCCCCC\",\n", - " \"tick\": \"#666666\",\n", - "}\n", - "\n", - "\n", - "def _style_axis(ax):\n", - " \"\"\"Apply shared cosmetic styling to a subplot axis.\"\"\"\n", - " ax.set_facecolor(\"white\")\n", - " ax.grid(True, alpha=0.4, color=_PALETTE[\"grid\"], linewidth=0.8)\n", - " for spine in (\"top\", \"right\"):\n", - " ax.spines[spine].set_visible(False)\n", - " ax.spines[\"left\"].set_color(_PALETTE[\"spine\"])\n", - " ax.spines[\"bottom\"].set_color(_PALETTE[\"spine\"])\n", - " ax.tick_params(colors=_PALETTE[\"tick\"], labelsize=9)\n", - "\n", - "\n", - "def _plot_line(ax, xs, ys, color, label, fill=True):\n", - " \"\"\"Plot a time series, gracefully skipping None values.\"\"\"\n", - " pts = [(x, y) for x, y in zip(xs, ys) if y is not None]\n", - " if not pts:\n", - " return\n", - " px, py = zip(*pts)\n", - " ax.plot(\n", - " px, py, color=color, linewidth=2.2,\n", - " marker=\"o\", markersize=4,\n", - " markerfacecolor=\"white\", markeredgewidth=1.8, markeredgecolor=color,\n", - " label=label, zorder=3,\n", - " )\n", - " if fill:\n", - " ax.fill_between(px, py, alpha=0.08, color=color)\n", - "\n", - "\n", - "def _plot_gpu_panel(ax, xs, history, colors, fallback_label):\n", - " \"\"\"Plot per-GPU time series with area fill.\"\"\"\n", - " if not history or not history[0]:\n", - " ax.text(\n", - " 0.5, 0.5, \"No GPU data\", transform=ax.transAxes,\n", - " ha=\"center\", va=\"center\", fontsize=11, color=\"#AAAAAA\",\n", - " )\n", - " return\n", - " n_gpus = max(len(snap) for snap in history)\n", - " for g in range(n_gpus):\n", - " vals = [snap[g] if g < len(snap) else 0 for snap in history]\n", - " c = colors[g % len(colors)]\n", - " label = f\"GPU {g}\" if n_gpus > 1 else fallback_label\n", - " ax.plot(xs[: len(vals)], vals, color=c, linewidth=2, label=label)\n", - " ax.fill_between(xs[: len(vals)], vals, alpha=0.08, color=c)\n", - " if n_gpus > 1:\n", - " ax.legend(fontsize=9, framealpha=0.9, edgecolor=\"#DDD\")\n", - "\n", - "\n", - "def _draw_dashboard(\n", - " elapsed_mins, val_losses, train_losses,\n", - " vram_history, util_history,\n", - " job_name, status_str, step_str, elapsed_str,\n", - "):\n", - " \"\"\"Render a live 1x3 training dashboard.\"\"\"\n", - " fig, axes = plt.subplots(1, 3, figsize=(20, 5.5))\n", - " fig.patch.set_facecolor(\"#FAFBFC\")\n", - "\n", - " fig.suptitle(\n", - " job_name, fontsize=15, fontweight=\"bold\",\n", - " color=_PALETTE[\"title\"], y=1.10,\n", - " )\n", - " fig.text(\n", - " 0.5, 1.01,\n", - " f\"{status_str} | {step_str} | {elapsed_str}\",\n", - " ha=\"center\", fontsize=13, color=_PALETTE[\"subtitle\"],\n", - " )\n", - "\n", - " for ax in axes:\n", - " _style_axis(ax)\n", - "\n", - " # -- Panel 1: Loss curves --\n", - " _plot_line(axes[0], elapsed_mins, val_losses, _PALETTE[\"val_loss\"], \"Val Loss\", fill=True)\n", - " _plot_line(axes[0], elapsed_mins, train_losses, _PALETTE[\"train_loss\"], \"Train Loss\", fill=False)\n", - " axes[0].set_title(\"Train/Validation Loss\", fontsize=13, fontweight=\"bold\", color=_PALETTE[\"title\"], pad=12)\n", - " axes[0].set_xlabel(\"Time (min)\", fontsize=10, color=\"#666\")\n", - " axes[0].set_ylabel(\"Loss\", fontsize=10, color=\"#666\")\n", - " if any(v is not None for v in val_losses + train_losses):\n", - " axes[0].legend(fontsize=9, framealpha=0.9, edgecolor=\"#DDD\")\n", - "\n", - " # -- Panel 2: GPU VRAM usage --\n", - " _plot_gpu_panel(axes[1], elapsed_mins, vram_history, _PALETTE[\"vram\"], \"VRAM\")\n", - " axes[1].set_title(\"GPU VRAM Usage\", fontsize=13, fontweight=\"bold\", color=_PALETTE[\"title\"], pad=12)\n", - " axes[1].set_xlabel(\"Time (min)\", fontsize=10, color=\"#666\")\n", - " axes[1].set_ylabel(\"Usage (%)\", fontsize=10, color=\"#666\")\n", - " axes[1].set_ylim(-2, 105)\n", - "\n", - " # -- Panel 3: GPU utilization --\n", - " _plot_gpu_panel(axes[2], elapsed_mins, util_history, _PALETTE[\"util\"], \"Utilization\")\n", - " axes[2].set_title(\"GPU Utilization\", fontsize=13, fontweight=\"bold\", color=_PALETTE[\"title\"], pad=12)\n", - " axes[2].set_xlabel(\"Time (min)\", fontsize=10, color=\"#666\")\n", - " axes[2].set_ylabel(\"Utilization (%)\", fontsize=10, color=\"#666\")\n", - " axes[2].set_ylim(-2, 105)\n", - "\n", - " plt.tight_layout(rect=[0, 0, 1, 0.98])\n", - " plt.show()" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### Monitor the Job Until Completion\n", - "\n", - "The cell below polls the job status every 10 seconds and renders a live dashboard with validation loss, GPU VRAM usage, and GPU utilization charts. The charts appear empty at first while the model and dataset download; training metrics and GPU activity populate after the finetuning step begins.\n", - "\n", - "> **Note:** This is additional code. You can also use the Weights & Biases or MLflow integrations." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import time\n", - "from typing import cast\n", - "from IPython.display import clear_output\n", - "from nemo_platform.types.shared import PlatformJobStatusResponse\n", - "\n", - "# Timeout set to 30 minutes to accommodate typical LoRA training duration for this dataset size.\n", - "# Actual training time will vary based on hardware, model size, and dataset complexity.\n", - "TIMEOUT_SECONDS = 30 * 60 # 30 minutes\n", - "VAL_LOSS_KEY = \"val_loss\"\n", - "TRAIN_LOSS_KEY = \"loss\"\n", - "\n", - "# ---------------------------------------------------------------------------\n", - "# Job polling with live dashboard\n", - "# ---------------------------------------------------------------------------\n", - "\n", - "def wait_for_job(\n", - " workspace: str,\n", - " job_name: str,\n", - " timeout: int = TIMEOUT_SECONDS,\n", - " poll_interval: int = 10,\n", - " val_loss_key: str = VAL_LOSS_KEY,\n", - " train_loss_key: str = TRAIN_LOSS_KEY,\n", - ") -> PlatformJobStatusResponse:\n", - " \"\"\"\n", - " Poll job status until completed, failed, cancelled, or timeout.\n", - " Displays a live dashboard with loss curves and GPU metrics.\n", - "\n", - " Args:\n", - " workspace: The workspace where the job is running.\n", - " job_name: The name of the job to monitor.\n", - " timeout: Maximum time to wait in seconds (default: 30 minutes).\n", - " poll_interval: Time between status checks in seconds (default: 10).\n", - "\n", - " Returns:\n", - " The final job status response.\n", - " \"\"\"\n", - " start_time = time.time()\n", - "\n", - " # Time-series accumulators required for plotting\n", - " elapsed_mins: list[float] = []\n", - " val_losses: list[float | None] = []\n", - " train_losses: list[float | None] = []\n", - " vram_history: list[list[float]] = []\n", - " util_history: list[list[float]] = []\n", - "\n", - " while True:\n", - " elapsed = time.time() - start_time\n", - " elapsed_min = elapsed / 60\n", - "\n", - " # Check for timeout\n", - " if elapsed > timeout:\n", - " error_message = f\"Timeout reached after {elapsed_min:.1f} minutes\"\n", - " print(f\"\\n{error_message}\")\n", - " print(\"Job did not complete within the timeout period.\")\n", - " raise Exception(error_message)\n", - "\n", - " status = client.jobs.get_status(name=job_name, workspace=workspace)\n", - "\n", - " # -- Extract training progress from nested steps structure --\n", - " step: int | None = None\n", - " max_steps: int | None = None\n", - " training_phase: str | None = None\n", - " val_loss: float | None = None\n", - " train_loss: float | None = None\n", - " current_step_name: str | None = None\n", - " current_step_phase: str | None = None\n", - "\n", - " for job_step in status.steps or []:\n", - " # Track the current active step name and phase for progress display\n", - " if job_step.tasks:\n", - " task = job_step.tasks[0]\n", - " td = task.status_details or {}\n", - " phase = cast(str, td.get(\"phase\", \"\"))\n", - " # Update current step if it's active or pending (not completed)\n", - " if job_step.status in (\"active\", \"pending\"):\n", - " current_step_name = job_step.name\n", - " current_step_phase = phase or \"started\"\n", - "\n", - " if job_step.name == \"customization-training-job\":\n", - " for task in job_step.tasks or []:\n", - " td = task.status_details or {}\n", - " step = cast(int, td[\"step\"]) if \"step\" in td else None\n", - " max_steps = cast(int, td[\"max_steps\"]) if \"max_steps\" in td else None\n", - " training_phase = cast(str, td[\"phase\"]) if \"phase\" in td else None\n", - " val_loss = float(td[val_loss_key]) if val_loss_key in td else None\n", - " train_loss = float(td[train_loss_key]) if train_loss_key in td else None\n", - " break\n", - " break\n", - "\n", - " # Fall back to top-level status_details\n", - " if status.status_details:\n", - " if val_loss is None and val_loss_key in status.status_details:\n", - " val_loss = float(status.status_details[val_loss_key])\n", - " if train_loss is None and train_loss_key in status.status_details:\n", - " train_loss = float(status.status_details[train_loss_key])\n", - "\n", - " # -- Collect GPU snapshot --\n", - " vram_pcts, util_pcts = _get_gpu_snapshot()\n", - "\n", - " # -- Append to accumulators used for the plots --\n", - " elapsed_mins.append(elapsed_min)\n", - " val_losses.append(val_loss)\n", - " train_losses.append(train_loss)\n", - " vram_history.append(vram_pcts)\n", - " util_history.append(util_pcts)\n", - "\n", - " # -- Build status strings --\n", - " status_str = f\"Status: {status.status}\"\n", - " if step is not None and max_steps is not None:\n", - " pct = step / max_steps * 100\n", - " step_str = f\"Step {step}/{max_steps} ({pct:.0f}%)\"\n", - " if training_phase:\n", - " step_str += f\" - {training_phase}\"\n", - " else:\n", - " if current_step_name and current_step_phase:\n", - " step_str = f\"{current_step_name} - {current_step_phase}\"\n", - " elif current_step_name:\n", - " step_str = f\"{current_step_name}\"\n", - " else:\n", - " step_str = \"Waiting for training to start...\"\n", - " elapsed_str = f\"Elapsed: {elapsed_min:.1f} min\"\n", - "\n", - " # -- Redraw dashboard --\n", - " clear_output(wait=True)\n", - " _draw_dashboard(\n", - " elapsed_mins, val_losses, train_losses,\n", - " vram_history, util_history,\n", - " job_name, status_str, step_str, elapsed_str,\n", - " )\n", - "\n", - " # -- Check terminal conditions --\n", - " if status.status.lower() == \"completed\":\n", - " # Redraw dashboard one final time with \"completed\" status\n", - " status_str = f\"Status: {status.status}\"\n", - " if step is not None and max_steps is not None:\n", - " step_str = f\"Step {max_steps}/{max_steps} (100%)\"\n", - " clear_output(wait=True)\n", - " _draw_dashboard(\n", - " elapsed_mins, val_losses, train_losses,\n", - " vram_history, util_history,\n", - " job_name, status_str, step_str, elapsed_str,\n", - " )\n", - " print(f\"\\nJob completed in {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n", - " return status\n", - " elif status.status.lower() in (\"failed\", \"cancelled\", \"error\"):\n", - " print(f\"\\nJob finished with status: {status.status}\")\n", - " print(f\"Total time elapsed: {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n", - "\n", - " # Print error details from the job level\n", - " if status.error_details:\n", - " error_msg = status.error_details.get(\"message\", \"\")\n", - " if error_msg:\n", - " print(f\"\\nError: {error_msg}\")\n", - "\n", - " # Find and print error details from the failed step/task\n", - " for job_step in status.steps or []:\n", - " if job_step.status == \"error\":\n", - " print(f\"\\nFailed step: {job_step.name}\")\n", - " if job_step.error_details:\n", - " step_error = job_step.error_details.get(\"message\", \"\")\n", - " if step_error:\n", - " print(f\"Step error: {step_error}\")\n", - " # Get error_stack from the failed task\n", - " for task in job_step.tasks or []:\n", - " if task.status == \"error\" and hasattr(task, \"error_stack\") and task.error_stack:\n", - " print(f\"\\nError stack trace:\\n{task.error_stack}\")\n", - " elif task.status == \"error\" and task.error_details:\n", - " task_error = task.error_details.get(\"message\", \"\")\n", - " if task_error:\n", - " print(f\"Task error: {task_error}\")\n", - " break\n", - "\n", - " raise Exception(f\"Job finished with status: {status.status}\")\n", - "\n", - " time.sleep(poll_interval)\n", - "\n", - "\n", - "# Wait for the job to complete\n", - "job_with_sequence_packing_status = wait_for_job(\n", - " workspace=\"default\",\n", - " job_name=job_with_sequence_packing.name,\n", - " timeout=TIMEOUT_SECONDS,\n", - ")\n", - "\n", - "print(f\"Validation loss: {job_with_sequence_packing_status.status_details['val_loss']:.2f}\")\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 7. Create LoRA Job without Sequence Packing\n", - "Create a customization job with sequence packing disabled. It's expected to take longer to complete." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "import uuid\n", - "\n", - "job_suffix = uuid.uuid4().hex[:4]\n", - "JOB_NAME = f\"my-sft-job-{job_suffix}\"\n", - "\n", - "job_spec_no_packing = CustomizationJobInputParam(\n", - " model=f\"default/{base_model.name}\",\n", - " dataset=f\"fileset://default/{DATASET_NAME}\",\n", - " training=SftTrainingParam(\n", - " type=\"sft\",\n", - " epochs=1,\n", - " batch_size=64,\n", - " learning_rate=0.00005,\n", - " max_seq_length=4096,\n", - " val_check_interval=0.1,\n", - " micro_batch_size=1,\n", - " sequence_packing=False,\n", - " peft=LoRaParamsParam(),\n", - " parallelism=ParallelismParamsParam(\n", - " num_gpus_per_node=1,\n", - " num_nodes=1,\n", - " tensor_parallel_size=1,\n", - " pipeline_parallel_size=1,\n", - " ),\n", - " )\n", - ")\n", - "\n", - "job_without_sequence_packing = client.customization.jobs.create(\n", - " name=JOB_NAME,\n", - " workspace=\"default\",\n", - " spec=job_spec_no_packing\n", - ")\n", - "\n", - "print(f\"Job ID: {job_without_sequence_packing.name}\")\n", - "print(f\"Output model: {job_without_sequence_packing.spec.output.name}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 8. Track Finetuning Progress for Job without Sequence Packing" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Wait for the training step to complete\n", - "job_without_sequence_packing_status = wait_for_job(\n", - " workspace=\"default\",\n", - " job_name=job_without_sequence_packing.name,\n", - " timeout=TIMEOUT_SECONDS\n", - ")\n", - "\n", - "print(f\"Validation loss: {job_without_sequence_packing_status.status_details['val_loss']:.2f}\")" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### 9. Compare Results\n", - "- Time to complete training should be significantly lower for the job that used sequence packing.\n", - "- The expected validation loss for both jobs should be similar.\n", - "- Sequence packed version should have a higher GPU utilization and higher GPU Memory Allocation." - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from nemo_platform.types.jobs import PlatformJobStep\n", - "from datetime import datetime\n", - "import pandas as pd\n", - "\n", - "STEP_NAME = \"customization-training-job\"\n", - "\n", - "def get_elapsed_time(step: PlatformJobStep) -> float:\n", - " \"\"\"Calculate elapsed time in seconds from step's created_at to updated_at.\"\"\"\n", - " created_at = datetime.fromisoformat(step.created_at.replace(\"Z\", \"+00:00\"))\n", - " updated_at = datetime.fromisoformat(step.updated_at.replace(\"Z\", \"+00:00\"))\n", - " return (updated_at - created_at).total_seconds()\n", - "\n", - "step_with_sequence_packing = client.jobs.steps.retrieve(\n", - " name=STEP_NAME,\n", - " workspace=\"default\",\n", - " job=job_with_sequence_packing.name,\n", - ")\n", - "\n", - "step_without_sequence_packing = client.jobs.steps.retrieve(\n", - " name=STEP_NAME,\n", - " workspace=\"default\",\n", - " job=job_without_sequence_packing.name,\n", - ")\n", - "\n", - "time_to_complete_with_sequence_packing = get_elapsed_time(step_with_sequence_packing)\n", - "time_to_complete_without_sequence_packing = get_elapsed_time(step_without_sequence_packing)\n", - "\n", - "# Display results as a table\n", - "results_df = pd.DataFrame({\n", - " \"Seq Packing Enabled\": [True, False],\n", - " \"Val Loss\": [\n", - " job_with_sequence_packing_status.status_details['val_loss'],\n", - " job_without_sequence_packing_status.status_details['val_loss']\n", - " ],\n", - " \"Training Step Time, sec\": [\n", - " time_to_complete_with_sequence_packing,\n", - " time_to_complete_without_sequence_packing\n", - " ]\n", - "})\n", - "\n", - "results_df.style.format({\"Val Loss\": \"{:.2f}\", \"Training Step Time, sec\": \"{:.0f}\"}).hide(axis='index')" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "#### Examples of Validation Loss\n", - "\n", - "The expected validation loss curves should match closely for both jobs.\n", - "![Validation loss comparison chart showing similar convergence patterns between sequence-packed and non-packed training runs over training steps](../_images/packed_vs_not_packed_val_loss.png)\n", - "\n", - "Sequence packed version should complete significantly faster.\n", - "![Runtime comparison chart demonstrating significantly reduced training time for sequence-packed job compared to non-packed baseline](../_images/runtime.png)\n", - "\n", - "#### GPU Utilization\n", - "Sequence packed version should have a higher GPU utilization.\n", - "![GPU utilization chart showing higher and more consistent GPU usage with sequence packing enabled throughout the training process](../_images/gpu_utilization.png)\n", - "\n", - "#### GPU Memory Allocation\n", - "Sequence packed version should have a higher GPU Memory Allocation.\n", - "![GPU memory allocation chart illustrating increased memory utilization efficiency with sequence packing enabled](../_images/gpu_memory.png)" - ] - } - ], - "metadata": { - "kernelspec": { - "display_name": ".venv", - "language": "python", - "name": "python3" - }, - "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 3 - }, - "file_extension": ".py", - "mimetype": "text/x-python", - "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython3", - "version": "3.11.14" + "cells": [ + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "\n", + "\n", + "\n", + "# Optimize for Tokens/GPU Throughput\n", + "\n", + "## About\n", + "Learn how to use the NeMo Platform Customizer to create a [LoRA](nemo-ms-about-concepts-customization) (Low-Rank Adaptation) customization job optimized for higher tokens/GPU throughput and lower runtime. \n", + "\n", + "**In this tutorial, you will:**\n", + "1. Fine-tune [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) on the SQuAD dataset using LoRA, with [sequence packing](nemo-ms-about-concepts-customization) enabled for one run and disabled for another.\n", + "2. Compare training runtime, GPU utilization, and memory allocation between the two runs.\n", + "3. Verify that validation loss remains comparable, confirming that sequence packing improves throughput without sacrificing model quality.\n", + "\n", + "> **Note:** While this tutorial demonstrates sequence packing with LoRA, the optimization is also available for all_weights (full) SFT customization jobs." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Prerequisites\n", + "\n", + "Before starting this tutorial, ensure you have:\n", + "\n", + "1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n", + "2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "## Quick Start\n", + "\n", + "### 1. Initialize SDK\n", + "\n", + "The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](../../get-started/quickstart.md) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable:\n", + "\n", + "```sh\n", + "export NMP_BASE_URL=\n", + "```" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "from nemo_platform import NeMoPlatform, ConflictError\n", + "\n", + "NMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\n", + "client = NeMoPlatform(\n", + " base_url=NMP_BASE_URL,\n", + " workspace=\"default\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 2. Create Dataset FileSet and Upload Training Data" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Install additional dependencies if they are not installed in your Python environment.\n", + "\n", + "The cell below automatically detects your environment and uses:\n", + "- `uv pip install` if you're in a uv-managed virtual environment\n", + "- `pip install` otherwise\n", + "\n", + "Required packages:\n", + "- `datasets` - Download the public [rajpurkar/squad](https://huggingface.co/datasets/rajpurkar/squad) dataset\n", + "- `pandas` - Compare job results in table format\n", + "- `matplotlib` - Plot live training metrics (loss curves, GPU utilization)\n", + "- `nvidia-ml-py` - Collect GPU VRAM and compute utilization metrics during training" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "vscode": { + "languageId": "shellscript" } + }, + "outputs": [], + "source": [ + "%%bash\n", + "if command -v uv >/dev/null 2>&1 && [ -n \"$VIRTUAL_ENV\" ]; then\n", + " uv pip install datasets pandas matplotlib nvidia-ml-py\n", + "else\n", + " pip install datasets pandas matplotlib nvidia-ml-py\n", + "fi" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### Download rajpurkar/squad Dataset\n", + "\n", + "SQuAD (Stanford Question Answering Dataset) is a reading comprehension dataset consisting of questions posed on Wikipedia articles, where the answer is a segment of text from the corresponding passage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "from pathlib import Path\n", + "from datasets import load_dataset, Dataset, DatasetDict\n", + "\n", + "# Configuration\n", + "SEED = 1234\n", + "DATASET_NAME = \"sft-dataset\"\n", + "\n", + "# Convert SQuAD format to prompt/completion format and save to JSONL\n", + "def convert_squad_to_sft_format(example):\n", + " \"\"\"Convert SQuAD format to prompt/completion format for SFT training.\"\"\"\n", + " prompt = f\"Context: {example['context']} Question: {example['question']} Answer:\"\n", + " completion = example[\"answers\"][\"text\"][0] # Take the first answer\n", + " return {\"prompt\": prompt, \"completion\": completion}\n", + "\n", + "# Load the SQuAD dataset from Hugging Face\n", + "print(\"Loading dataset rajpurkar/squad\")\n", + "ds = load_dataset(\"rajpurkar/squad\")\n", + "if not isinstance(ds, DatasetDict):\n", + " raise ValueError(\"Dataset does not contain expected splits\")\n", + "\n", + "print(\"Loaded dataset\")\n", + "\n", + "# For the purpose of this tutorial, we'll use a subset of the dataset\n", + "# We use a reduced dataset size (3000 training/300 validation samples) to keep tutorial runtime manageable\n", + "# while still demonstrating the performance benefits of sequence packing. The larger the dataset,\n", + "# the better the model will perform but the longer the training will take.\n", + "training_size = 3000\n", + "validation_size = 300\n", + "DATASET_PATH = Path(DATASET_NAME).absolute()\n", + "\n", + "# Get training split and verify it's a Dataset (not IterableDataset)\n", + "train_dataset = ds[\"train\"]\n", + "validation_dataset = ds[\"validation\"]\n", + "assert isinstance(train_dataset, Dataset), \"Expected Dataset type\"\n", + "assert isinstance(validation_dataset, Dataset), \"Expected Dataset type\"\n", + "\n", + "# Select subsets and save to JSONL files\n", + "training_ds = train_dataset.select(range(training_size))\n", + "validation_ds = validation_dataset.select(range(validation_size))\n", + "\n", + "# Transform to SFT format (prompt/completion)\n", + "training_ds = training_ds.map(convert_squad_to_sft_format, remove_columns=training_ds.column_names)\n", + "validation_ds = validation_ds.map(convert_squad_to_sft_format, remove_columns=validation_ds.column_names)\n", + "\n", + "# Create directory if it doesn't exist\n", + "# Note: This will create a local 'sft-dataset/' directory with training.jsonl and validation.jsonl files\n", + "os.makedirs(DATASET_PATH, exist_ok=True)\n", + "\n", + "# Save subsets to JSONL files\n", + "training_ds.to_json(f\"{DATASET_PATH}/training.jsonl\")\n", + "validation_ds.to_json(f\"{DATASET_PATH}/validation.jsonl\")\n", + "\n", + "print(f\"Saved training.jsonl with {len(training_ds)} rows\")\n", + "print(f\"Saved validation.jsonl with {len(validation_ds)} rows\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Create fileset to store SFT training data\n", + "\n", + "try:\n", + " client.files.filesets.create(\n", + " workspace=\"default\",\n", + " name=DATASET_NAME,\n", + " description=\"SFT training data\"\n", + " )\n", + " print(f\"Created fileset: {DATASET_NAME}\")\n", + "except ConflictError:\n", + " print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n", + "\n", + "# Upload training data files individually to ensure correct structure\n", + "client.files.upload(\n", + " local_path=f\"{DATASET_PATH}/\", # Trailing slash uploads directory contents to fileset root\n", + " remote_path=\"\",\n", + " fileset=DATASET_NAME,\n", + " workspace=\"default\"\n", + ")\n", + "\n", + "# Validate training data is uploaded correctly\n", + "print(\"Training data:\")\n", + "print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 3. Secrets Setup\n", + "\n", + "If you plan to use NGC or HuggingFace models, you will need to configure authentication:\n", + "\n", + "- **NGC models** (`ngc://` URIs): Requires NGC API key\n", + "- **HuggingFace models** (`hf://` URIs): Requires HF token for gated/private models\n", + "\n", + "\n", + "Configure these as secrets in your platform. Refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md) for detailed instructions.\n", + "\n", + "Get your credentials to access base models:\n", + "- [NGC API Key](https://ngc.nvidia.com/) (Setup → Generate API Key)\n", + "- [HuggingFace Token](https://huggingface.co/settings/tokens) (Create token with Read access)\n", + "\n", + "\n", + "---\n", + "\n", + "#### Quick Setup Example\n", + "\n", + "This tutorial uses the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model from HuggingFace. Ensure that you have sufficient permissions to download the model. If you cannot access the files on the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) Hugging Face page, request access.\n", + "\n", + "**HuggingFace Authentication:**\n", + "- For gated models (Llama, Gemma), you must provide a HuggingFace token via the `token_secret` parameter\n", + "- Get your token from [HuggingFace Settings](https://huggingface.co/settings/tokens) (requires Read access)\n", + "- Accept the model's terms on the HuggingFace model page before using it. Example: [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main)\n", + "- For public models, you can omit the `token_secret` parameter when creating a fileset for the model in the next step." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Export the HF_TOKEN and NGC_API_KEY environment variables if they are not already set\n", + "HF_TOKEN = os.getenv(\"HF_TOKEN\")\n", + "NGC_API_KEY = os.getenv(\"NGC_API_KEY\")\n", + "\n", + "\n", + "def create_or_get_secret(name: str, value: str | None, label: str):\n", + " if not value:\n", + " raise ValueError(f\"{label} environment variable is not set. Set it and try again.\")\n", + " try:\n", + " secret = client.secrets.create(\n", + " name=name,\n", + " workspace=\"default\",\n", + " value=value,\n", + " )\n", + " print(f\"Created secret: {name}\")\n", + " return secret\n", + " except ConflictError:\n", + " print(f\"Secret '{name}' already exists, continuing...\")\n", + " return client.secrets.retrieve(name=name, workspace=\"default\")\n", + "\n", + "\n", + "# Create HuggingFace token secret\n", + "hf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\n", + "print(\"HF_TOKEN secret:\")\n", + "print(hf_secret.model_dump_json(indent=2))\n", + "\n", + "# Create NGC API key secret\n", + "# Uncomment the line below if you have NGC API Key and want to finetune NGC models\n", + "# ngc_api_key = create_or_get_secret(\"ngc-api-key\", NGC_API_KEY, \"NGC_API_KEY\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 4. Create Base Model FileSet\n", + "\n", + "Create a fileset pointing to the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model on HuggingFace. This step creates a pointer to the model on Hugging Face and does not download it. The model is downloaded at job creation time.\n", + "\n", + "Note: for public models, you can omit the `token_secret` parameter when creating a model fileset." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import time\n", + "from nemo_platform.types.files import HuggingfaceStorageConfigParam\n", + "\n", + "HF_REPO_ID = \"meta-llama/Llama-3.2-1B-Instruct\"\n", + "MODEL_NAME = \"llama-3-2-1b-base\"\n", + "\n", + "# Ensure you have a HuggingFace token secret created\n", + "# Create a fileset pointing to the desired HuggingFace model\n", + "try:\n", + " base_model_fs = client.files.filesets.create(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " description=\"Llama 3.2 1B base model from HuggingFace\",\n", + " storage=HuggingfaceStorageConfigParam(\n", + " type=\"huggingface\",\n", + " # repo_id is the full model name from Hugging Face\n", + " repo_id=HF_REPO_ID,\n", + " repo_type=\"model\",\n", + " # we use the secret created in the previous step\n", + " token_secret=hf_secret.name\n", + " )\n", + " )\n", + " print(f\"Created base model fileset: {MODEL_NAME}\")\n", + "except ConflictError:\n", + " print(f\"Base model fileset already exists. Skipping creation.\")\n", + " base_model_fs = client.files.filesets.retrieve(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " )\n", + "\n", + "# Create the Model Entity representation.\n", + "try:\n", + " base_model = client.models.create(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " fileset=f\"default/{MODEL_NAME}\",\n", + " )\n", + " print(f\"Created Model Entity: {MODEL_NAME}\")\n", + "except ConflictError:\n", + " print(f\"Base model already exists. Updating fileset if different.\")\n", + " base_model = client.models.update(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " fileset=f\"default/{MODEL_NAME}\",\n", + " )\n", + "\n", + "print(f\"\\nBase model fileset: fileset://default/{base_model.name}\")\n", + "print(\"Base model fileset files list:\")\n", + "print(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=\"default\").data], indent=2))\n", + "\n", + "# Wait for ModelSpec to be populated from the checkpoint\n", + "print(\"\\nWaiting for ModelSpec to be populated...\")\n", + "SPEC_TIMEOUT_SECONDS = 120\n", + "spec_start = time.time()\n", + "while not base_model.spec:\n", + " if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n", + " raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n", + " time.sleep(2)\n", + " base_model = client.models.retrieve(\n", + " workspace=\"default\",\n", + " name=MODEL_NAME,\n", + " )\n", + "\n", + "print(f\"ModelSpec populated: {base_model.spec}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 5. Create LoRA Job with Sequence Packing\n", + "Create a LoRA customization job with **sequence packing** enabled via `AutomodelJobInput` (`batch.sequence_packing=True`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import uuid\n", + "from nemo_automodel_plugin.schema import AutomodelJobInput\n", + "\n", + "SEQUENCE_PACKING_ENABLED = True\n", + "\n", + "job_suffix = uuid.uuid4().hex[:4]\n", + "JOB_NAME = f\"packing-job-{job_suffix}\"\n", + "PACK_OUTPUT_NAME = f\"packing-out-{job_suffix}\"\n", + "\n", + "spec = AutomodelJobInput(\n", + " model=f\"default/{base_model.name}\",\n", + " dataset={\"training\": f\"default/{DATASET_NAME}\"},\n", + " training={\n", + " \"training_type\": \"sft\",\n", + " \"finetuning_type\": \"lora\",\n", + " \"max_seq_length\": 4096,\n", + " },\n", + " schedule={\"epochs\": 1, \"val_check_interval\": 0.1},\n", + " batch={\n", + " \"global_batch_size\": 64,\n", + " \"micro_batch_size\": 1,\n", + " \"sequence_packing\": SEQUENCE_PACKING_ENABLED,\n", + " },\n", + " optimizer={\"learning_rate\": 5e-5},\n", + " parallelism={\"num_gpus_per_node\": 1},\n", + " output={\"name\": PACK_OUTPUT_NAME},\n", + ")\n", + "\n", + "job_with_sequence_packing = client.customization.automodel.jobs.create(\n", + " spec=spec, workspace=\"default\", name=JOB_NAME\n", + ")\n", + "\n", + "print(f\"Submitted job: {job_with_sequence_packing.job.name}\")\n", + "print(f\"Output adapter: {PACK_OUTPUT_NAME}\")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 6. Track Finetuning Progress\n", + "\n", + "A training job contains multiple steps: \n", + "- Model and dataset downloading\n", + "- Finetuning where LoRA adapter weights are trained\n", + "- Creating a fileset entry for the finetuned model\n", + "- Finetuned weights uploading\n", + "\n", + "The elapsed time printed below reflects progress of the entire job. We compare the time taken by the finetuning step for both jobs in the last section of this tutorial." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### Define Helper Functions" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "\n", + "# Helpers to draw GPU VRAM Utilization and Validation Loss\n", + "import matplotlib.pyplot as plt\n", + "try:\n", + " import pynvml\n", + " _PYNVML_AVAILABLE = True\n", + "except ImportError:\n", + " _PYNVML_AVAILABLE = False\n", + " print(\"Note: Install nvidia-ml-py ('pip install nvidia-ml-py' or 'uv pip install nvidia-ml-py') to enable live GPU metrics.\")\n", + "\n", + "# ---------------------------------------------------------------------------\n", + "# GPU metrics collection (nvidia-ml-py; import name is pynvml)\n", + "# ---------------------------------------------------------------------------\n", + "\n", + "def _get_gpu_snapshot() -> tuple[list[float], list[float]]:\n", + " \"\"\"Return (vram_usage_pcts, compute_util_pcts) for each GPU.\"\"\"\n", + " if not _PYNVML_AVAILABLE:\n", + " return [], []\n", + " pynvml.nvmlInit()\n", + " try:\n", + " vram, util = [], []\n", + " for i in range(pynvml.nvmlDeviceGetCount()):\n", + " h = pynvml.nvmlDeviceGetHandleByIndex(i)\n", + " mem = pynvml.nvmlDeviceGetMemoryInfo(h)\n", + " rates = pynvml.nvmlDeviceGetUtilizationRates(h)\n", + " vram.append(int(mem.used) / int(mem.total) * 100)\n", + " util.append(float(rates.gpu))\n", + " return vram, util\n", + " finally:\n", + " pynvml.nvmlShutdown()\n", + "\n", + "\n", + "# ---------------------------------------------------------------------------\n", + "# Dashboard drawing helpers\n", + "# ---------------------------------------------------------------------------\n", + "\n", + "_PALETTE = {\n", + " \"val_loss\": \"#E74C3C\",\n", + " \"train_loss\": \"#F39C12\",\n", + " \"vram\": [\"#3498DB\", \"#9B59B6\", \"#1ABC9C\", \"#E67E22\"],\n", + " \"util\": [\"#2ECC71\", \"#E74C3C\", \"#3498DB\", \"#F1C40F\"],\n", + " \"grid\": \"#ECECEC\",\n", + " \"title\": \"#2C3E50\",\n", + " \"subtitle\": \"#7F8C8D\",\n", + " \"spine\": \"#CCCCCC\",\n", + " \"tick\": \"#666666\",\n", + "}\n", + "\n", + "\n", + "def _style_axis(ax):\n", + " \"\"\"Apply shared cosmetic styling to a subplot axis.\"\"\"\n", + " ax.set_facecolor(\"white\")\n", + " ax.grid(True, alpha=0.4, color=_PALETTE[\"grid\"], linewidth=0.8)\n", + " for spine in (\"top\", \"right\"):\n", + " ax.spines[spine].set_visible(False)\n", + " ax.spines[\"left\"].set_color(_PALETTE[\"spine\"])\n", + " ax.spines[\"bottom\"].set_color(_PALETTE[\"spine\"])\n", + " ax.tick_params(colors=_PALETTE[\"tick\"], labelsize=9)\n", + "\n", + "\n", + "def _plot_line(ax, xs, ys, color, label, fill=True):\n", + " \"\"\"Plot a time series, gracefully skipping None values.\"\"\"\n", + " pts = [(x, y) for x, y in zip(xs, ys) if y is not None]\n", + " if not pts:\n", + " return\n", + " px, py = zip(*pts)\n", + " ax.plot(\n", + " px, py, color=color, linewidth=2.2,\n", + " marker=\"o\", markersize=4,\n", + " markerfacecolor=\"white\", markeredgewidth=1.8, markeredgecolor=color,\n", + " label=label, zorder=3,\n", + " )\n", + " if fill:\n", + " ax.fill_between(px, py, alpha=0.08, color=color)\n", + "\n", + "\n", + "def _plot_gpu_panel(ax, xs, history, colors, fallback_label):\n", + " \"\"\"Plot per-GPU time series with area fill.\"\"\"\n", + " if not history or not history[0]:\n", + " ax.text(\n", + " 0.5, 0.5, \"No GPU data\", transform=ax.transAxes,\n", + " ha=\"center\", va=\"center\", fontsize=11, color=\"#AAAAAA\",\n", + " )\n", + " return\n", + " n_gpus = max(len(snap) for snap in history)\n", + " for g in range(n_gpus):\n", + " vals = [snap[g] if g < len(snap) else 0 for snap in history]\n", + " c = colors[g % len(colors)]\n", + " label = f\"GPU {g}\" if n_gpus > 1 else fallback_label\n", + " ax.plot(xs[: len(vals)], vals, color=c, linewidth=2, label=label)\n", + " ax.fill_between(xs[: len(vals)], vals, alpha=0.08, color=c)\n", + " if n_gpus > 1:\n", + " ax.legend(fontsize=9, framealpha=0.9, edgecolor=\"#DDD\")\n", + "\n", + "\n", + "def _draw_dashboard(\n", + " elapsed_mins, val_losses, train_losses,\n", + " vram_history, util_history,\n", + " job_name, status_str, step_str, elapsed_str,\n", + "):\n", + " \"\"\"Render a live 1x3 training dashboard.\"\"\"\n", + " fig, axes = plt.subplots(1, 3, figsize=(20, 5.5))\n", + " fig.patch.set_facecolor(\"#FAFBFC\")\n", + "\n", + " fig.suptitle(\n", + " job_name, fontsize=15, fontweight=\"bold\",\n", + " color=_PALETTE[\"title\"], y=1.10,\n", + " )\n", + " fig.text(\n", + " 0.5, 1.01,\n", + " f\"{status_str} | {step_str} | {elapsed_str}\",\n", + " ha=\"center\", fontsize=13, color=_PALETTE[\"subtitle\"],\n", + " )\n", + "\n", + " for ax in axes:\n", + " _style_axis(ax)\n", + "\n", + " # -- Panel 1: Loss curves --\n", + " _plot_line(axes[0], elapsed_mins, val_losses, _PALETTE[\"val_loss\"], \"Val Loss\", fill=True)\n", + " _plot_line(axes[0], elapsed_mins, train_losses, _PALETTE[\"train_loss\"], \"Train Loss\", fill=False)\n", + " axes[0].set_title(\"Train/Validation Loss\", fontsize=13, fontweight=\"bold\", color=_PALETTE[\"title\"], pad=12)\n", + " axes[0].set_xlabel(\"Time (min)\", fontsize=10, color=\"#666\")\n", + " axes[0].set_ylabel(\"Loss\", fontsize=10, color=\"#666\")\n", + " if any(v is not None for v in val_losses + train_losses):\n", + " axes[0].legend(fontsize=9, framealpha=0.9, edgecolor=\"#DDD\")\n", + "\n", + " # -- Panel 2: GPU VRAM usage --\n", + " _plot_gpu_panel(axes[1], elapsed_mins, vram_history, _PALETTE[\"vram\"], \"VRAM\")\n", + " axes[1].set_title(\"GPU VRAM Usage\", fontsize=13, fontweight=\"bold\", color=_PALETTE[\"title\"], pad=12)\n", + " axes[1].set_xlabel(\"Time (min)\", fontsize=10, color=\"#666\")\n", + " axes[1].set_ylabel(\"Usage (%)\", fontsize=10, color=\"#666\")\n", + " axes[1].set_ylim(-2, 105)\n", + "\n", + " # -- Panel 3: GPU utilization --\n", + " _plot_gpu_panel(axes[2], elapsed_mins, util_history, _PALETTE[\"util\"], \"Utilization\")\n", + " axes[2].set_title(\"GPU Utilization\", fontsize=13, fontweight=\"bold\", color=_PALETTE[\"title\"], pad=12)\n", + " axes[2].set_xlabel(\"Time (min)\", fontsize=10, color=\"#666\")\n", + " axes[2].set_ylabel(\"Utilization (%)\", fontsize=10, color=\"#666\")\n", + " axes[2].set_ylim(-2, 105)\n", + "\n", + " plt.tight_layout(rect=[0, 0, 1, 0.98])\n", + " plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### Monitor the Job Until Completion\n", + "\n", + "The cell below polls the job status every 10 seconds and renders a live dashboard with validation loss, GPU VRAM usage, and GPU utilization charts. The charts appear empty at first while the model and dataset download; training metrics and GPU activity populate after the finetuning step begins.\n", + "\n", + "> **Note:** This is additional code. You can also use the Weights & Biases or MLflow integrations." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import time\n", + "from typing import cast\n", + "from IPython.display import clear_output\n", + "from nemo_platform.types.shared import PlatformJobStatusResponse\n", + "\n", + "# Timeout set to 30 minutes to accommodate typical LoRA training duration for this dataset size.\n", + "# Actual training time will vary based on hardware, model size, and dataset complexity.\n", + "TIMEOUT_SECONDS = 30 * 60 # 30 minutes\n", + "VAL_LOSS_KEY = \"val_loss\"\n", + "TRAIN_LOSS_KEY = \"loss\"\n", + "\n", + "# ---------------------------------------------------------------------------\n", + "# Job polling with live dashboard\n", + "# ---------------------------------------------------------------------------\n", + "\n", + "def wait_for_job(\n", + " workspace: str,\n", + " job_name: str,\n", + " timeout: int = TIMEOUT_SECONDS,\n", + " poll_interval: int = 10,\n", + " val_loss_key: str = VAL_LOSS_KEY,\n", + " train_loss_key: str = TRAIN_LOSS_KEY,\n", + ") -> PlatformJobStatusResponse:\n", + " \"\"\"\n", + " Poll job status until completed, failed, cancelled, or timeout.\n", + " Displays a live dashboard with loss curves and GPU metrics.\n", + "\n", + " Args:\n", + " workspace: The workspace where the job is running.\n", + " job_name: The name of the job to monitor.\n", + " timeout: Maximum time to wait in seconds (default: 30 minutes).\n", + " poll_interval: Time between status checks in seconds (default: 10).\n", + "\n", + " Returns:\n", + " The final job status response.\n", + " \"\"\"\n", + " start_time = time.time()\n", + "\n", + " # Time-series accumulators required for plotting\n", + " elapsed_mins: list[float] = []\n", + " val_losses: list[float | None] = []\n", + " train_losses: list[float | None] = []\n", + " vram_history: list[list[float]] = []\n", + " util_history: list[list[float]] = []\n", + "\n", + " while True:\n", + " elapsed = time.time() - start_time\n", + " elapsed_min = elapsed / 60\n", + "\n", + " # Check for timeout\n", + " if elapsed > timeout:\n", + " error_message = f\"Timeout reached after {elapsed_min:.1f} minutes\"\n", + " print(f\"\\n{error_message}\")\n", + " print(\"Job did not complete within the timeout period.\")\n", + " raise Exception(error_message)\n", + "\n", + " status = client.jobs.get_status(name=job_name, workspace=workspace)\n", + "\n", + " # -- Extract training progress from nested steps structure --\n", + " step: int | None = None\n", + " max_steps: int | None = None\n", + " training_phase: str | None = None\n", + " val_loss: float | None = None\n", + " train_loss: float | None = None\n", + " current_step_name: str | None = None\n", + " current_step_phase: str | None = None\n", + "\n", + " for job_step in status.steps or []:\n", + " # Track the current active step name and phase for progress display\n", + " if job_step.tasks:\n", + " task = job_step.tasks[0]\n", + " td = task.status_details or {}\n", + " phase = cast(str, td.get(\"phase\", \"\"))\n", + " # Update current step if it's active or pending (not completed)\n", + " if job_step.status in (\"active\", \"pending\"):\n", + " current_step_name = job_step.name\n", + " current_step_phase = phase or \"started\"\n", + "\n", + " if job_step.name == \"training\":\n", + " for task in job_step.tasks or []:\n", + " td = task.status_details or {}\n", + " step = cast(int, td[\"step\"]) if \"step\" in td else None\n", + " max_steps = cast(int, td[\"max_steps\"]) if \"max_steps\" in td else None\n", + " training_phase = cast(str, td[\"phase\"]) if \"phase\" in td else None\n", + " raw_val_loss = td.get(val_loss_key)\n", + " val_loss = float(raw_val_loss) if raw_val_loss is not None else None\n", + " raw_train_loss = td.get(train_loss_key)\n", + " train_loss = float(raw_train_loss) if raw_train_loss is not None else None\n", + " break\n", + " break\n", + "\n", + " if val_loss is None:\n", + " raw_val_loss = (status.status_details or {}).get(val_loss_key)\n", + " val_loss = float(raw_val_loss) if raw_val_loss is not None else None\n", + " if train_loss is None:\n", + " raw_train_loss = (status.status_details or {}).get(train_loss_key)\n", + " train_loss = float(raw_train_loss) if raw_train_loss is not None else None\n", + "\n", + " # -- Collect GPU snapshot --\n", + " vram_pcts, util_pcts = _get_gpu_snapshot()\n", + "\n", + " # -- Append to accumulators used for the plots --\n", + " elapsed_mins.append(elapsed_min)\n", + " val_losses.append(val_loss)\n", + " train_losses.append(train_loss)\n", + " vram_history.append(vram_pcts)\n", + " util_history.append(util_pcts)\n", + "\n", + " # -- Build status strings --\n", + " status_str = f\"Status: {status.status}\"\n", + " if step is not None and max_steps is not None:\n", + " pct = step / max_steps * 100\n", + " step_str = f\"Step {step}/{max_steps} ({pct:.0f}%)\"\n", + " if training_phase:\n", + " step_str += f\" - {training_phase}\"\n", + " else:\n", + " if current_step_name and current_step_phase:\n", + " step_str = f\"{current_step_name} - {current_step_phase}\"\n", + " elif current_step_name:\n", + " step_str = f\"{current_step_name}\"\n", + " else:\n", + " step_str = \"Waiting for training to start...\"\n", + " elapsed_str = f\"Elapsed: {elapsed_min:.1f} min\"\n", + "\n", + " # -- Redraw dashboard --\n", + " clear_output(wait=True)\n", + " _draw_dashboard(\n", + " elapsed_mins, val_losses, train_losses,\n", + " vram_history, util_history,\n", + " job_name, status_str, step_str, elapsed_str,\n", + " )\n", + "\n", + " # -- Check terminal conditions --\n", + " if status.status.lower() == \"completed\":\n", + " # Redraw dashboard one final time with \"completed\" status\n", + " status_str = f\"Status: {status.status}\"\n", + " if step is not None and max_steps is not None:\n", + " step_str = f\"Step {max_steps}/{max_steps} (100%)\"\n", + " clear_output(wait=True)\n", + " _draw_dashboard(\n", + " elapsed_mins, val_losses, train_losses,\n", + " vram_history, util_history,\n", + " job_name, status_str, step_str, elapsed_str,\n", + " )\n", + " print(f\"\\nJob completed in {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n", + " return status\n", + " elif status.status.lower() in (\"failed\", \"cancelled\", \"error\"):\n", + " print(f\"\\nJob finished with status: {status.status}\")\n", + " print(f\"Total time elapsed: {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n", + "\n", + " # Print error details from the job level\n", + " if status.error_details:\n", + " error_msg = status.error_details.get(\"message\", \"\")\n", + " if error_msg:\n", + " print(f\"\\nError: {error_msg}\")\n", + "\n", + " # Find and print error details from the failed step/task\n", + " for job_step in status.steps or []:\n", + " if job_step.status == \"error\":\n", + " print(f\"\\nFailed step: {job_step.name}\")\n", + " if job_step.error_details:\n", + " step_error = job_step.error_details.get(\"message\", \"\")\n", + " if step_error:\n", + " print(f\"Step error: {step_error}\")\n", + " # Get error_stack from the failed task\n", + " for task in job_step.tasks or []:\n", + " if task.status == \"error\" and hasattr(task, \"error_stack\") and task.error_stack:\n", + " print(f\"\\nError stack trace:\\n{task.error_stack}\")\n", + " elif task.status == \"error\" and task.error_details:\n", + " task_error = task.error_details.get(\"message\", \"\")\n", + " if task_error:\n", + " print(f\"Task error: {task_error}\")\n", + " break\n", + "\n", + " raise Exception(f\"Job finished with status: {status.status}\")\n", + "\n", + " time.sleep(poll_interval)\n", + "\n", + "\n", + "# Wait for the job to complete\n", + "job_with_sequence_packing_status = wait_for_job(\n", + " workspace=\"default\",\n", + " job_name=job_with_sequence_packing.job.name,\n", + " timeout=TIMEOUT_SECONDS,\n", + ")\n", + "\n", + "packed_val_loss = (job_with_sequence_packing_status.status_details or {}).get(\"val_loss\")\n", + "if packed_val_loss is not None:\n", + " print(f\"Validation loss: {float(packed_val_loss):.2f}\")\n", + "else:\n", + " print(\"Validation loss: not reported in job status\")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 7. Create LoRA Job without Sequence Packing\n", + "Create a second Automodel LoRA job with `batch.sequence_packing=False` for comparison." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "import uuid\n", + "from nemo_automodel_plugin.schema import AutomodelJobInput\n", + "\n", + "job_suffix = uuid.uuid4().hex[:4]\n", + "JOB_NAME = f\"no-packing-job-{job_suffix}\"\n", + "NO_PACK_OUTPUT_NAME = f\"no-packing-out-{job_suffix}\"\n", + "\n", + "spec = AutomodelJobInput(\n", + " model=f\"default/{base_model.name}\",\n", + " dataset={\"training\": f\"default/{DATASET_NAME}\"},\n", + " training={\n", + " \"training_type\": \"sft\",\n", + " \"finetuning_type\": \"lora\",\n", + " \"max_seq_length\": 4096,\n", + " },\n", + " schedule={\"epochs\": 1, \"val_check_interval\": 0.1},\n", + " batch={\n", + " \"global_batch_size\": 64,\n", + " \"micro_batch_size\": 1,\n", + " \"sequence_packing\": False,\n", + " },\n", + " optimizer={\"learning_rate\": 5e-5},\n", + " parallelism={\"num_gpus_per_node\": 1},\n", + " output={\"name\": NO_PACK_OUTPUT_NAME},\n", + ")\n", + "\n", + "job_without_sequence_packing = client.customization.automodel.jobs.create(\n", + " spec=spec, workspace=\"default\", name=JOB_NAME\n", + ")\n", + "\n", + "print(f\"Submitted job: {job_without_sequence_packing.job.name}\")\n", + "print(f\"Output adapter: {NO_PACK_OUTPUT_NAME}\")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 8. Track Finetuning Progress for Job without Sequence Packing" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Wait for the training step to complete\n", + "job_without_sequence_packing_status = wait_for_job(\n", + " workspace=\"default\",\n", + " job_name=job_without_sequence_packing.job.name,\n", + " timeout=TIMEOUT_SECONDS\n", + ")\n", + "\n", + "no_pack_val_loss = (job_without_sequence_packing_status.status_details or {}).get(\"val_loss\")\n", + "if no_pack_val_loss is not None:\n", + " print(f\"Validation loss: {float(no_pack_val_loss):.2f}\")\n", + "else:\n", + " print(\"Validation loss: not reported in job status\")" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### 9. Compare Results\n", + "- Time to complete training should be significantly lower for the job that used sequence packing.\n", + "- The expected validation loss for both jobs should be similar.\n", + "- Sequence packed version should have a higher GPU utilization and higher GPU Memory Allocation." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "from nemo_platform.types.jobs import PlatformJobStep\n", + "from datetime import datetime\n", + "import pandas as pd\n", + "\n", + "STEP_NAME = \"training\"\n", + "\n", + "def get_elapsed_time(step: PlatformJobStep) -> float:\n", + " \"\"\"Calculate elapsed time in seconds from step's created_at to updated_at.\"\"\"\n", + " created_at = datetime.fromisoformat(step.created_at.replace(\"Z\", \"+00:00\"))\n", + " updated_at = datetime.fromisoformat(step.updated_at.replace(\"Z\", \"+00:00\"))\n", + " return (updated_at - created_at).total_seconds()\n", + "\n", + "step_with_sequence_packing = client.jobs.steps.retrieve(\n", + " name=STEP_NAME,\n", + " workspace=\"default\",\n", + " job=job_with_sequence_packing.job.name,\n", + ")\n", + "\n", + "step_without_sequence_packing = client.jobs.steps.retrieve(\n", + " name=STEP_NAME,\n", + " workspace=\"default\",\n", + " job=job_without_sequence_packing.job.name,\n", + ")\n", + "\n", + "time_to_complete_with_sequence_packing = get_elapsed_time(step_with_sequence_packing)\n", + "time_to_complete_without_sequence_packing = get_elapsed_time(step_without_sequence_packing)\n", + "\n", + "# Display results as a table\n", + "results_df = pd.DataFrame({\n", + " \"Seq Packing Enabled\": [True, False],\n", + " \"Val Loss\": [\n", + " (job_with_sequence_packing_status.status_details or {}).get(\"val_loss\"),\n", + " (job_without_sequence_packing_status.status_details or {}).get(\"val_loss\"),\n", + " ],\n", + " \"Training Step Time, sec\": [\n", + " time_to_complete_with_sequence_packing,\n", + " time_to_complete_without_sequence_packing\n", + " ]\n", + "})\n", + "\n", + "results_df.style.format({\"Val Loss\": \"{:.2f}\", \"Training Step Time, sec\": \"{:.0f}\"}).hide(axis='index')" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### Examples of Validation Loss\n", + "\n", + "The expected validation loss curves should match closely for both jobs.\n", + "![Validation loss comparison chart showing similar convergence patterns between sequence-packed and non-packed training runs over training steps](../_images/packed_vs_not_packed_val_loss.png)\n", + "\n", + "Sequence packed version should complete significantly faster.\n", + "![Runtime comparison chart demonstrating significantly reduced training time for sequence-packed job compared to non-packed baseline](../_images/runtime.png)\n", + "\n", + "#### GPU Utilization\n", + "Sequence packed version should have a higher GPU utilization.\n", + "![GPU utilization chart showing higher and more consistent GPU usage with sequence packing enabled throughout the training process](../_images/gpu_utilization.png)\n", + "\n", + "#### GPU Memory Allocation\n", + "Sequence packed version should have a higher GPU Memory Allocation.\n", + "![GPU memory allocation chart illustrating increased memory utilization efficiency with sequence packing enabled](../_images/gpu_memory.png)" + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": ".venv", + "language": "python", + "name": "python3" }, - "nbformat": 4, - "nbformat_minor": 2 + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.11.14" + } + }, + "nbformat": 4, + "nbformat_minor": 2 } diff --git a/docs/customizer/tutorials/optimize-throughput.mdx b/docs/customizer/tutorials/optimize-throughput.mdx index 7cb9e80b75..20dc00fdf5 100644 --- a/docs/customizer/tutorials/optimize-throughput.mdx +++ b/docs/customizer/tutorials/optimize-throughput.mdx @@ -3,7 +3,809 @@ title: "Optimize for Tokens/GPU Throughput" description: "" --- - +[Run in Google Colab](https://colab.research.google.com/github/NVIDIA-NeMo/nemo-platform/blob/main/docs/customizer/tutorials/optimize-throughput.ipynb) + +# Optimize for Tokens/GPU Throughput + +## About +Learn how to use the NeMo Platform Customizer to create a [LoRA](/documentation/customizer-reference/customization-concepts#nemo-ms-about-concepts-customization) (Low-Rank Adaptation) customization job optimized for higher tokens/GPU throughput and lower runtime. + +**In this tutorial, you will:** +1. Fine-tune [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) on the SQuAD dataset using LoRA, with [sequence packing](/documentation/customizer-reference/customization-concepts#nemo-ms-about-concepts-customization) enabled for one run and disabled for another. +2. Compare training runtime, GPU utilization, and memory allocation between the two runs. +3. Verify that validation loss remains comparable, confirming that sequence packing improves throughput without sacrificing model quality. + +> **Note:** While this tutorial demonstrates sequence packing with LoRA, the optimization is also available for all_weights (full) SFT customization jobs. + +## Prerequisites + +Before starting this tutorial, ensure you have: + +1. **Completed the [Quickstart](/documentation/get-started)** to install and deploy NeMo Platform locally +2. **Installed the Python SDK** (PyPI wrapper: `pip install "nemo-platform[all]"`; source checkout: run `make bootstrap` from the repository root) + +## Quick Start + +### 1. Initialize SDK + +The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](/documentation/get-started) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable: + +```sh +export NMP_BASE_URL= +``` + +```python +import json +import os +from nemo_platform import NeMoPlatform, ConflictError + +NMP_BASE_URL = os.environ.get("NMP_BASE_URL", "http://localhost:8080") +client = NeMoPlatform( + base_url=NMP_BASE_URL, + workspace="default" +) +``` + +### 2. Create Dataset FileSet and Upload Training Data + +Install additional dependencies if they are not installed in your Python environment. + +The cell below automatically detects your environment and uses: +- `uv pip install` if you're in a uv-managed virtual environment +- `pip install` otherwise + +Required packages: +- `datasets` - Download the public [rajpurkar/squad](https://huggingface.co/datasets/rajpurkar/squad) dataset +- `pandas` - Compare job results in table format +- `matplotlib` - Plot live training metrics (loss curves, GPU utilization) +- `nvidia-ml-py` - Collect GPU VRAM and compute utilization metrics during training + +```sh +%%bash +if command -v uv >/dev/null 2>&1 && [ -n "$VIRTUAL_ENV" ]; then + uv pip install datasets pandas matplotlib nvidia-ml-py +else + pip install datasets pandas matplotlib nvidia-ml-py +fi +``` + +#### Download rajpurkar/squad Dataset + +SQuAD (Stanford Question Answering Dataset) is a reading comprehension dataset consisting of questions posed on Wikipedia articles, where the answer is a segment of text from the corresponding passage. + +```python +import json +import os +from pathlib import Path +from datasets import load_dataset, Dataset, DatasetDict + +# Configuration +SEED = 1234 +DATASET_NAME = "sft-dataset" + +# Convert SQuAD format to prompt/completion format and save to JSONL +def convert_squad_to_sft_format(example): + """Convert SQuAD format to prompt/completion format for SFT training.""" + prompt = f"Context: {example['context']} Question: {example['question']} Answer:" + completion = example["answers"]["text"][0] # Take the first answer + return {"prompt": prompt, "completion": completion} + +# Load the SQuAD dataset from Hugging Face +print("Loading dataset rajpurkar/squad") +ds = load_dataset("rajpurkar/squad") +if not isinstance(ds, DatasetDict): + raise ValueError("Dataset does not contain expected splits") + +print("Loaded dataset") + +# For the purpose of this tutorial, we'll use a subset of the dataset +# We use a reduced dataset size (3000 training/300 validation samples) to keep tutorial runtime manageable +# while still demonstrating the performance benefits of sequence packing. The larger the dataset, +# the better the model will perform but the longer the training will take. +training_size = 3000 +validation_size = 300 +DATASET_PATH = Path(DATASET_NAME).absolute() + +# Get training split and verify it's a Dataset (not IterableDataset) +train_dataset = ds["train"] +validation_dataset = ds["validation"] +assert isinstance(train_dataset, Dataset), "Expected Dataset type" +assert isinstance(validation_dataset, Dataset), "Expected Dataset type" + +# Select subsets and save to JSONL files +training_ds = train_dataset.select(range(training_size)) +validation_ds = validation_dataset.select(range(validation_size)) + +# Transform to SFT format (prompt/completion) +training_ds = training_ds.map(convert_squad_to_sft_format, remove_columns=training_ds.column_names) +validation_ds = validation_ds.map(convert_squad_to_sft_format, remove_columns=validation_ds.column_names) + +# Create directory if it doesn't exist +# Note: This will create a local 'sft-dataset/' directory with training.jsonl and validation.jsonl files +os.makedirs(DATASET_PATH, exist_ok=True) + +# Save subsets to JSONL files +training_ds.to_json(f"{DATASET_PATH}/training.jsonl") +validation_ds.to_json(f"{DATASET_PATH}/validation.jsonl") + +print(f"Saved training.jsonl with {len(training_ds)} rows") +print(f"Saved validation.jsonl with {len(validation_ds)} rows") +``` + +```python +# Create fileset to store SFT training data + +try: + client.files.filesets.create( + workspace="default", + name=DATASET_NAME, + description="SFT training data" + ) + print(f"Created fileset: {DATASET_NAME}") +except ConflictError: + print(f"Fileset '{DATASET_NAME}' already exists, continuing...") + +# Upload training data files individually to ensure correct structure +client.files.upload( + local_path=f"{DATASET_PATH}/", # Trailing slash uploads directory contents to fileset root + remote_path="", + fileset=DATASET_NAME, + workspace="default" +) + +# Validate training data is uploaded correctly +print("Training data:") +print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2)) +``` + +### 3. Secrets Setup + +If you plan to use NGC or HuggingFace models, you will need to configure authentication: + +- **NGC models** (`ngc://` URIs): Requires NGC API key +- **HuggingFace models** (`hf://` URIs): Requires HF token for gated/private models + + +Configure these as secrets in your platform. Refer to [Managing Secrets](/documentation/get-started/core-concepts/manage-secrets) for detailed instructions. + +Get your credentials to access base models: +- [NGC API Key](https://ngc.nvidia.com/) (Setup → Generate API Key) +- [HuggingFace Token](https://huggingface.co/settings/tokens) (Create token with Read access) + + +--- + +#### Quick Setup Example + +This tutorial uses the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model from HuggingFace. Ensure that you have sufficient permissions to download the model. If you cannot access the files on the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) Hugging Face page, request access. + +**HuggingFace Authentication:** +- For gated models (Llama, Gemma), you must provide a HuggingFace token via the `token_secret` parameter +- Get your token from [HuggingFace Settings](https://huggingface.co/settings/tokens) (requires Read access) +- Accept the model's terms on the HuggingFace model page before using it. Example: [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) +- For public models, you can omit the `token_secret` parameter when creating a fileset for the model in the next step. + +```python +# Export the HF_TOKEN and NGC_API_KEY environment variables if they are not already set +HF_TOKEN = os.getenv("HF_TOKEN") +NGC_API_KEY = os.getenv("NGC_API_KEY") + + +def create_or_get_secret(name: str, value: str | None, label: str): + if not value: + raise ValueError(f"{label} environment variable is not set. Set it and try again.") + try: + secret = client.secrets.create( + name=name, + workspace="default", + value=value, + ) + print(f"Created secret: {name}") + return secret + except ConflictError: + print(f"Secret '{name}' already exists, continuing...") + return client.secrets.retrieve(name=name, workspace="default") + + +# Create HuggingFace token secret +hf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN") +print("HF_TOKEN secret:") +print(hf_secret.model_dump_json(indent=2)) + +# Create NGC API key secret +# Uncomment the line below if you have NGC API Key and want to finetune NGC models +# ngc_api_key = create_or_get_secret("ngc-api-key", NGC_API_KEY, "NGC_API_KEY") +``` + +### 4. Create Base Model FileSet + +Create a fileset pointing to the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model on HuggingFace. This step creates a pointer to the model on Hugging Face and does not download it. The model is downloaded at job creation time. + +Note: for public models, you can omit the `token_secret` parameter when creating a model fileset. + +```python +import time +from nemo_platform.types.files import HuggingfaceStorageConfigParam + +HF_REPO_ID = "meta-llama/Llama-3.2-1B-Instruct" +MODEL_NAME = "llama-3-2-1b-base" + +# Ensure you have a HuggingFace token secret created +# Create a fileset pointing to the desired HuggingFace model +try: + base_model_fs = client.files.filesets.create( + workspace="default", + name=MODEL_NAME, + description="Llama 3.2 1B base model from HuggingFace", + storage=HuggingfaceStorageConfigParam( + type="huggingface", + # repo_id is the full model name from Hugging Face + repo_id=HF_REPO_ID, + repo_type="model", + # we use the secret created in the previous step + token_secret=hf_secret.name + ) + ) + print(f"Created base model fileset: {MODEL_NAME}") +except ConflictError: + print(f"Base model fileset already exists. Skipping creation.") + base_model_fs = client.files.filesets.retrieve( + workspace="default", + name=MODEL_NAME, + ) + +# Create the Model Entity representation. +try: + base_model = client.models.create( + workspace="default", + name=MODEL_NAME, + fileset=f"default/{MODEL_NAME}", + ) + print(f"Created Model Entity: {MODEL_NAME}") +except ConflictError: + print(f"Base model already exists. Updating fileset if different.") + base_model = client.models.update( + workspace="default", + name=MODEL_NAME, + fileset=f"default/{MODEL_NAME}", + ) + +print(f"\nBase model fileset: fileset://default/{base_model.name}") +print("Base model fileset files list:") +print(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace="default").data], indent=2)) + +# Wait for ModelSpec to be populated from the checkpoint +print("\nWaiting for ModelSpec to be populated...") +SPEC_TIMEOUT_SECONDS = 120 +spec_start = time.time() +while not base_model.spec: + if time.time() - spec_start > SPEC_TIMEOUT_SECONDS: + raise TimeoutError(f"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds") + time.sleep(2) + base_model = client.models.retrieve( + workspace="default", + name=MODEL_NAME, + ) + +print(f"ModelSpec populated: {base_model.spec}") +``` + +### 5. Create LoRA Job with Sequence Packing +Create a LoRA customization job with **sequence packing** enabled via `AutomodelJobInput` (`batch.sequence_packing=True`). + +```python +import uuid +from nemo_automodel_plugin.schema import AutomodelJobInput + +SEQUENCE_PACKING_ENABLED = True + +job_suffix = uuid.uuid4().hex[:4] +JOB_NAME = f"packing-job-{job_suffix}" +PACK_OUTPUT_NAME = f"packing-out-{job_suffix}" + +spec = AutomodelJobInput( + model=f"default/{base_model.name}", + dataset={"training": f"default/{DATASET_NAME}"}, + training={ + "training_type": "sft", + "finetuning_type": "lora", + "max_seq_length": 4096, + }, + schedule={"epochs": 1, "val_check_interval": 0.1}, + batch={ + "global_batch_size": 64, + "micro_batch_size": 1, + "sequence_packing": SEQUENCE_PACKING_ENABLED, + }, + optimizer={"learning_rate": 5e-5}, + parallelism={"num_gpus_per_node": 1}, + output={"name": PACK_OUTPUT_NAME}, +) + +job_with_sequence_packing = client.customization.automodel.jobs.create( + spec=spec, workspace="default", name=JOB_NAME +) + +print(f"Submitted job: {job_with_sequence_packing.job.name}") +print(f"Output adapter: {PACK_OUTPUT_NAME}") + +``` + +### 6. Track Finetuning Progress + +A training job contains multiple steps: +- Model and dataset downloading +- Finetuning where LoRA adapter weights are trained +- Creating a fileset entry for the finetuned model +- Finetuned weights uploading + +The elapsed time printed below reflects progress of the entire job. We compare the time taken by the finetuning step for both jobs in the last section of this tutorial. + +#### Define Helper Functions + + +```python +# Helpers to draw GPU VRAM Utilization and Validation Loss +import matplotlib.pyplot as plt +try: + import pynvml + _PYNVML_AVAILABLE = True +except ImportError: + _PYNVML_AVAILABLE = False + print("Note: Install nvidia-ml-py ('pip install nvidia-ml-py' or 'uv pip install nvidia-ml-py') to enable live GPU metrics.") + +# --------------------------------------------------------------------------- +# GPU metrics collection (nvidia-ml-py; import name is pynvml) +# --------------------------------------------------------------------------- + +def _get_gpu_snapshot() -> tuple[list[float], list[float]]: + """Return (vram_usage_pcts, compute_util_pcts) for each GPU.""" + if not _PYNVML_AVAILABLE: + return [], [] + pynvml.nvmlInit() + try: + vram, util = [], [] + for i in range(pynvml.nvmlDeviceGetCount()): + h = pynvml.nvmlDeviceGetHandleByIndex(i) + mem = pynvml.nvmlDeviceGetMemoryInfo(h) + rates = pynvml.nvmlDeviceGetUtilizationRates(h) + vram.append(int(mem.used) / int(mem.total) * 100) + util.append(float(rates.gpu)) + return vram, util + finally: + pynvml.nvmlShutdown() + + +# --------------------------------------------------------------------------- +# Dashboard drawing helpers +# --------------------------------------------------------------------------- + +_PALETTE = { + "val_loss": "#E74C3C", + "train_loss": "#F39C12", + "vram": ["#3498DB", "#9B59B6", "#1ABC9C", "#E67E22"], + "util": ["#2ECC71", "#E74C3C", "#3498DB", "#F1C40F"], + "grid": "#ECECEC", + "title": "#2C3E50", + "subtitle": "#7F8C8D", + "spine": "#CCCCCC", + "tick": "#666666", +} + + +def _style_axis(ax): + """Apply shared cosmetic styling to a subplot axis.""" + ax.set_facecolor("white") + ax.grid(True, alpha=0.4, color=_PALETTE["grid"], linewidth=0.8) + for spine in ("top", "right"): + ax.spines[spine].set_visible(False) + ax.spines["left"].set_color(_PALETTE["spine"]) + ax.spines["bottom"].set_color(_PALETTE["spine"]) + ax.tick_params(colors=_PALETTE["tick"], labelsize=9) + + +def _plot_line(ax, xs, ys, color, label, fill=True): + """Plot a time series, gracefully skipping None values.""" + pts = [(x, y) for x, y in zip(xs, ys) if y is not None] + if not pts: + return + px, py = zip(*pts) + ax.plot( + px, py, color=color, linewidth=2.2, + marker="o", markersize=4, + markerfacecolor="white", markeredgewidth=1.8, markeredgecolor=color, + label=label, zorder=3, + ) + if fill: + ax.fill_between(px, py, alpha=0.08, color=color) + + +def _plot_gpu_panel(ax, xs, history, colors, fallback_label): + """Plot per-GPU time series with area fill.""" + if not history or not history[0]: + ax.text( + 0.5, 0.5, "No GPU data", transform=ax.transAxes, + ha="center", va="center", fontsize=11, color="#AAAAAA", + ) + return + n_gpus = max(len(snap) for snap in history) + for g in range(n_gpus): + vals = [snap[g] if g < len(snap) else 0 for snap in history] + c = colors[g % len(colors)] + label = f"GPU {g}" if n_gpus > 1 else fallback_label + ax.plot(xs[: len(vals)], vals, color=c, linewidth=2, label=label) + ax.fill_between(xs[: len(vals)], vals, alpha=0.08, color=c) + if n_gpus > 1: + ax.legend(fontsize=9, framealpha=0.9, edgecolor="#DDD") + + +def _draw_dashboard( + elapsed_mins, val_losses, train_losses, + vram_history, util_history, + job_name, status_str, step_str, elapsed_str, +): + """Render a live 1x3 training dashboard.""" + fig, axes = plt.subplots(1, 3, figsize=(20, 5.5)) + fig.patch.set_facecolor("#FAFBFC") + + fig.suptitle( + job_name, fontsize=15, fontweight="bold", + color=_PALETTE["title"], y=1.10, + ) + fig.text( + 0.5, 1.01, + f"{status_str} | {step_str} | {elapsed_str}", + ha="center", fontsize=13, color=_PALETTE["subtitle"], + ) + + for ax in axes: + _style_axis(ax) + + # -- Panel 1: Loss curves -- + _plot_line(axes[0], elapsed_mins, val_losses, _PALETTE["val_loss"], "Val Loss", fill=True) + _plot_line(axes[0], elapsed_mins, train_losses, _PALETTE["train_loss"], "Train Loss", fill=False) + axes[0].set_title("Train/Validation Loss", fontsize=13, fontweight="bold", color=_PALETTE["title"], pad=12) + axes[0].set_xlabel("Time (min)", fontsize=10, color="#666") + axes[0].set_ylabel("Loss", fontsize=10, color="#666") + if any(v is not None for v in val_losses + train_losses): + axes[0].legend(fontsize=9, framealpha=0.9, edgecolor="#DDD") + + # -- Panel 2: GPU VRAM usage -- + _plot_gpu_panel(axes[1], elapsed_mins, vram_history, _PALETTE["vram"], "VRAM") + axes[1].set_title("GPU VRAM Usage", fontsize=13, fontweight="bold", color=_PALETTE["title"], pad=12) + axes[1].set_xlabel("Time (min)", fontsize=10, color="#666") + axes[1].set_ylabel("Usage (%)", fontsize=10, color="#666") + axes[1].set_ylim(-2, 105) + + # -- Panel 3: GPU utilization -- + _plot_gpu_panel(axes[2], elapsed_mins, util_history, _PALETTE["util"], "Utilization") + axes[2].set_title("GPU Utilization", fontsize=13, fontweight="bold", color=_PALETTE["title"], pad=12) + axes[2].set_xlabel("Time (min)", fontsize=10, color="#666") + axes[2].set_ylabel("Utilization (%)", fontsize=10, color="#666") + axes[2].set_ylim(-2, 105) + + plt.tight_layout(rect=[0, 0, 1, 0.98]) + plt.show() +``` + +#### Monitor the Job Until Completion + +The cell below polls the job status every 10 seconds and renders a live dashboard with validation loss, GPU VRAM usage, and GPU utilization charts. The charts appear empty at first while the model and dataset download; training metrics and GPU activity populate after the finetuning step begins. + +> **Note:** This is additional code. You can also use the Weights & Biases or MLflow integrations. + +```python +import time +from typing import cast +from IPython.display import clear_output +from nemo_platform.types.shared import PlatformJobStatusResponse + +# Timeout set to 30 minutes to accommodate typical LoRA training duration for this dataset size. +# Actual training time will vary based on hardware, model size, and dataset complexity. +TIMEOUT_SECONDS = 30 * 60 # 30 minutes +VAL_LOSS_KEY = "val_loss" +TRAIN_LOSS_KEY = "loss" + +# --------------------------------------------------------------------------- +# Job polling with live dashboard +# --------------------------------------------------------------------------- + +def wait_for_job( + workspace: str, + job_name: str, + timeout: int = TIMEOUT_SECONDS, + poll_interval: int = 10, + val_loss_key: str = VAL_LOSS_KEY, + train_loss_key: str = TRAIN_LOSS_KEY, +) -> PlatformJobStatusResponse: + """ + Poll job status until completed, failed, cancelled, or timeout. + Displays a live dashboard with loss curves and GPU metrics. + + Args: + workspace: The workspace where the job is running. + job_name: The name of the job to monitor. + timeout: Maximum time to wait in seconds (default: 30 minutes). + poll_interval: Time between status checks in seconds (default: 10). + + Returns: + The final job status response. + """ + start_time = time.time() + + # Time-series accumulators required for plotting + elapsed_mins: list[float] = [] + val_losses: list[float | None] = [] + train_losses: list[float | None] = [] + vram_history: list[list[float]] = [] + util_history: list[list[float]] = [] + + while True: + elapsed = time.time() - start_time + elapsed_min = elapsed / 60 + + # Check for timeout + if elapsed > timeout: + error_message = f"Timeout reached after {elapsed_min:.1f} minutes" + print(f"\n{error_message}") + print("Job did not complete within the timeout period.") + raise Exception(error_message) + + status = client.jobs.get_status(name=job_name, workspace=workspace) + + # -- Extract training progress from nested steps structure -- + step: int | None = None + max_steps: int | None = None + training_phase: str | None = None + val_loss: float | None = None + train_loss: float | None = None + current_step_name: str | None = None + current_step_phase: str | None = None + + for job_step in status.steps or []: + # Track the current active step name and phase for progress display + if job_step.tasks: + task = job_step.tasks[0] + td = task.status_details or {} + phase = cast(str, td.get("phase", "")) + # Update current step if it's active or pending (not completed) + if job_step.status in ("active", "pending"): + current_step_name = job_step.name + current_step_phase = phase or "started" + + if job_step.name == "training": + for task in job_step.tasks or []: + td = task.status_details or {} + step = cast(int, td["step"]) if "step" in td else None + max_steps = cast(int, td["max_steps"]) if "max_steps" in td else None + training_phase = cast(str, td["phase"]) if "phase" in td else None + raw_val_loss = td.get(val_loss_key) + val_loss = float(raw_val_loss) if raw_val_loss is not None else None + raw_train_loss = td.get(train_loss_key) + train_loss = float(raw_train_loss) if raw_train_loss is not None else None + break + break + + if val_loss is None: + raw_val_loss = (status.status_details or {}).get(val_loss_key) + val_loss = float(raw_val_loss) if raw_val_loss is not None else None + if train_loss is None: + raw_train_loss = (status.status_details or {}).get(train_loss_key) + train_loss = float(raw_train_loss) if raw_train_loss is not None else None + + # -- Collect GPU snapshot -- + vram_pcts, util_pcts = _get_gpu_snapshot() + + # -- Append to accumulators used for the plots -- + elapsed_mins.append(elapsed_min) + val_losses.append(val_loss) + train_losses.append(train_loss) + vram_history.append(vram_pcts) + util_history.append(util_pcts) + + # -- Build status strings -- + status_str = f"Status: {status.status}" + if step is not None and max_steps is not None: + pct = step / max_steps * 100 + step_str = f"Step {step}/{max_steps} ({pct:.0f}%)" + if training_phase: + step_str += f" - {training_phase}" + else: + if current_step_name and current_step_phase: + step_str = f"{current_step_name} - {current_step_phase}" + elif current_step_name: + step_str = f"{current_step_name}" + else: + step_str = "Waiting for training to start..." + elapsed_str = f"Elapsed: {elapsed_min:.1f} min" + + # -- Redraw dashboard -- + clear_output(wait=True) + _draw_dashboard( + elapsed_mins, val_losses, train_losses, + vram_history, util_history, + job_name, status_str, step_str, elapsed_str, + ) + + # -- Check terminal conditions -- + if status.status.lower() == "completed": + # Redraw dashboard one final time with "completed" status + status_str = f"Status: {status.status}" + if step is not None and max_steps is not None: + step_str = f"Step {max_steps}/{max_steps} (100%)" + clear_output(wait=True) + _draw_dashboard( + elapsed_mins, val_losses, train_losses, + vram_history, util_history, + job_name, status_str, step_str, elapsed_str, + ) + print(f"\nJob completed in {elapsed_min:.1f} minutes ({elapsed:.0f}s)") + return status + elif status.status.lower() in ("failed", "cancelled", "error"): + print(f"\nJob finished with status: {status.status}") + print(f"Total time elapsed: {elapsed_min:.1f} minutes ({elapsed:.0f}s)") + + # Print error details from the job level + if status.error_details: + error_msg = status.error_details.get("message", "") + if error_msg: + print(f"\nError: {error_msg}") + + # Find and print error details from the failed step/task + for job_step in status.steps or []: + if job_step.status == "error": + print(f"\nFailed step: {job_step.name}") + if job_step.error_details: + step_error = job_step.error_details.get("message", "") + if step_error: + print(f"Step error: {step_error}") + # Get error_stack from the failed task + for task in job_step.tasks or []: + if task.status == "error" and hasattr(task, "error_stack") and task.error_stack: + print(f"\nError stack trace:\n{task.error_stack}") + elif task.status == "error" and task.error_details: + task_error = task.error_details.get("message", "") + if task_error: + print(f"Task error: {task_error}") + break + + raise Exception(f"Job finished with status: {status.status}") + + time.sleep(poll_interval) + + +# Wait for the job to complete +job_with_sequence_packing_status = wait_for_job( + workspace="default", + job_name=job_with_sequence_packing.job.name, + timeout=TIMEOUT_SECONDS, +) + +packed_val_loss = (job_with_sequence_packing_status.status_details or {}).get("val_loss") +if packed_val_loss is not None: + print(f"Validation loss: {float(packed_val_loss):.2f}") +else: + print("Validation loss: not reported in job status") + +``` + +### 7. Create LoRA Job without Sequence Packing +Create a second Automodel LoRA job with `batch.sequence_packing=False` for comparison. + +```python +import uuid +from nemo_automodel_plugin.schema import AutomodelJobInput + +job_suffix = uuid.uuid4().hex[:4] +JOB_NAME = f"no-packing-job-{job_suffix}" +NO_PACK_OUTPUT_NAME = f"no-packing-out-{job_suffix}" + +spec = AutomodelJobInput( + model=f"default/{base_model.name}", + dataset={"training": f"default/{DATASET_NAME}"}, + training={ + "training_type": "sft", + "finetuning_type": "lora", + "max_seq_length": 4096, + }, + schedule={"epochs": 1, "val_check_interval": 0.1}, + batch={ + "global_batch_size": 64, + "micro_batch_size": 1, + "sequence_packing": False, + }, + optimizer={"learning_rate": 5e-5}, + parallelism={"num_gpus_per_node": 1}, + output={"name": NO_PACK_OUTPUT_NAME}, +) + +job_without_sequence_packing = client.customization.automodel.jobs.create( + spec=spec, workspace="default", name=JOB_NAME +) + +print(f"Submitted job: {job_without_sequence_packing.job.name}") +print(f"Output adapter: {NO_PACK_OUTPUT_NAME}") + +``` + +### 8. Track Finetuning Progress for Job without Sequence Packing + +```python +# Wait for the training step to complete +job_without_sequence_packing_status = wait_for_job( + workspace="default", + job_name=job_without_sequence_packing.job.name, + timeout=TIMEOUT_SECONDS +) + +no_pack_val_loss = (job_without_sequence_packing_status.status_details or {}).get("val_loss") +if no_pack_val_loss is not None: + print(f"Validation loss: {float(no_pack_val_loss):.2f}") +else: + print("Validation loss: not reported in job status") +``` + +### 9. Compare Results +- Time to complete training should be significantly lower for the job that used sequence packing. +- The expected validation loss for both jobs should be similar. +- Sequence packed version should have a higher GPU utilization and higher GPU Memory Allocation. + +```python +from nemo_platform.types.jobs import PlatformJobStep +from datetime import datetime +import pandas as pd + +STEP_NAME = "training" + +def get_elapsed_time(step: PlatformJobStep) -> float: + """Calculate elapsed time in seconds from step's created_at to updated_at.""" + created_at = datetime.fromisoformat(step.created_at.replace("Z", "+00:00")) + updated_at = datetime.fromisoformat(step.updated_at.replace("Z", "+00:00")) + return (updated_at - created_at).total_seconds() + +step_with_sequence_packing = client.jobs.steps.retrieve( + name=STEP_NAME, + workspace="default", + job=job_with_sequence_packing.job.name, +) + +step_without_sequence_packing = client.jobs.steps.retrieve( + name=STEP_NAME, + workspace="default", + job=job_without_sequence_packing.job.name, +) + +time_to_complete_with_sequence_packing = get_elapsed_time(step_with_sequence_packing) +time_to_complete_without_sequence_packing = get_elapsed_time(step_without_sequence_packing) + +# Display results as a table +results_df = pd.DataFrame({ + "Seq Packing Enabled": [True, False], + "Val Loss": [ + (job_with_sequence_packing_status.status_details or {}).get("val_loss"), + (job_without_sequence_packing_status.status_details or {}).get("val_loss"), + ], + "Training Step Time, sec": [ + time_to_complete_with_sequence_packing, + time_to_complete_without_sequence_packing + ] +}) + +results_df.style.format({"Val Loss": "{:.2f}", "Training Step Time, sec": "{:.0f}"}).hide(axis='index') +``` + +#### Examples of Validation Loss + +The expected validation loss curves should match closely for both jobs. +![Validation loss comparison chart showing similar convergence patterns between sequence-packed and non-packed training runs over training steps](../_images/packed_vs_not_packed_val_loss.png) + +Sequence packed version should complete significantly faster. +![Runtime comparison chart demonstrating significantly reduced training time for sequence-packed job compared to non-packed baseline](../_images/runtime.png) + +#### GPU Utilization +Sequence packed version should have a higher GPU utilization. +![GPU utilization chart showing higher and more consistent GPU usage with sequence packing enabled throughout the training process](../_images/gpu_utilization.png) + +#### GPU Memory Allocation +Sequence packed version should have a higher GPU Memory Allocation. +![GPU memory allocation chart illustrating increased memory utilization efficiency with sequence packing enabled](../_images/gpu_memory.png) diff --git a/docs/customizer/tutorials/sft-customization-job.ipynb b/docs/customizer/tutorials/sft-customization-job.ipynb index 925e4346e2..0dc0515ad2 100644 --- a/docs/customizer/tutorials/sft-customization-job.ipynb +++ b/docs/customizer/tutorials/sft-customization-job.ipynb @@ -78,9 +78,7 @@ }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ "import json\n", "import os\n", @@ -91,7 +89,9 @@ " base_url=NMP_BASE_URL,\n", " workspace=\"default\"\n", ")" - ] + ], + "execution_count": null, + "outputs": [] }, { "cell_type": "markdown", @@ -200,9 +200,7 @@ }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ "from pathlib import Path\n", "from datasets import load_dataset, DatasetDict\n", @@ -266,13 +264,13 @@ " sample = json.loads(first_line)\n", " print(f\"Prompt: {sample['prompt'][:200]}...\")\n", " print(f\"Completion: {sample['completion']}\")" - ] + ], + "execution_count": null, + "outputs": [] }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ "# Create fileset to store SFT training data\n", "DATASET_NAME = \"sft-dataset\"\n", @@ -289,7 +287,7 @@ "\n", "# Upload training data files individually to ensure correct structure\n", "client.files.upload(\n", - " local_path=DATASET_PATH, # Local directory with your JSONL files\n", + " local_path=f\"{DATASET_PATH}/\", # Trailing slash uploads directory contents to fileset root\n", " remote_path=\"\",\n", " fileset=DATASET_NAME,\n", " workspace=\"default\"\n", @@ -298,7 +296,9 @@ "# Validate training data is uploaded correctly\n", "print(\"Training data:\")\n", "print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))" - ] + ], + "execution_count": null, + "outputs": [] }, { "cell_type": "markdown", @@ -334,9 +334,7 @@ }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ "# Export the HF_TOKEN and NGC_API_KEY environment variables if they are not already set\n", "HF_TOKEN = os.getenv(\"HF_TOKEN\")\n", @@ -367,7 +365,9 @@ "# Create NGC API key secret\n", "# Uncomment the line below if you have NGC API Key and want to finetune NGC models\n", "# ngc_api_key = create_or_get_secret(\"ngc-api-key\", NGC_API_KEY, \"NGC_API_KEY\")" - ] + ], + "execution_count": null, + "outputs": [] }, { "cell_type": "markdown", @@ -382,9 +382,7 @@ }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ "import time\n", "\n", @@ -451,14 +449,16 @@ " )\n", "\n", "print(f\"ModelSpec populated: {base_model.spec}\")" - ] + ], + "execution_count": null, + "outputs": [] }, { "cell_type": "markdown", "metadata": {}, "source": [ "### 6. Create SFT Finetuning Job\n", - "Create a customization job with an inline target referencing the base model and dataset filesets created in previous steps." + "Create a customization job to fine-tune all model weights using the **Automodel** backend and `AutomodelJobInput`." ] }, { @@ -477,47 +477,45 @@ }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ "import uuid\n", - "from nemo_platform.types.customization import (\n", - " CustomizationJobInputParam,\n", - " SftTrainingParam,\n", - " ParallelismParamsParam,\n", - ")\n", + "from nemo_automodel_plugin.schema import AutomodelJobInput\n", "\n", "job_suffix = uuid.uuid4().hex[:4]\n", "\n", "JOB_NAME = f\"my-sft-job-{job_suffix}\"\n", + "OUTPUT_NAME = f\"sft-model-{job_suffix}\"\n", + "\n", + "spec = AutomodelJobInput(\n", + " model=f\"default/{base_model.name}\",\n", + " dataset={\"training\": f\"default/{DATASET_NAME}\"},\n", + " training={\n", + " \"training_type\": \"sft\",\n", + " \"finetuning_type\": \"all_weights\",\n", + " \"max_seq_length\": 2048,\n", + " },\n", + " schedule={\"epochs\": 2},\n", + " batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n", + " optimizer={\"learning_rate\": 5e-5},\n", + " parallelism={\n", + " \"num_gpus_per_node\": 1,\n", + " \"num_nodes\": 1,\n", + " \"tensor_parallel_size\": 1,\n", + " \"pipeline_parallel_size\": 1,\n", + " },\n", + " output={\"name\": OUTPUT_NAME},\n", + ")\n", "\n", - "job = client.customization.jobs.create(\n", - " name=JOB_NAME,\n", - " workspace=\"default\",\n", - " spec=CustomizationJobInputParam(\n", - " model=f\"default/{base_model.name}\",\n", - " dataset=f\"fileset://default/{DATASET_NAME}\",\n", - " training=SftTrainingParam(\n", - " type=\"sft\",\n", - " epochs=2,\n", - " batch_size=64,\n", - " learning_rate=0.00005,\n", - " max_seq_length=2048,\n", - " micro_batch_size=1,\n", - " parallelism=ParallelismParamsParam(\n", - " num_gpus_per_node=1,\n", - " num_nodes=1,\n", - " tensor_parallel_size=1,\n", - " pipeline_parallel_size=1,\n", - " ),\n", - " ),\n", - " )\n", + "job = client.customization.automodel.jobs.create(\n", + " spec=spec, workspace=\"default\", name=JOB_NAME\n", ")\n", "\n", - "print(f\"Job ID: {job.name}\")\n", - "print(f\"Output model: {job.spec.output.name}\")" - ] + "print(f\"Submitted job: {job.job.name}\")\n", + "print(f\"Output model: {OUTPUT_NAME}\")\n" + ], + "execution_count": null, + "outputs": [] }, { "cell_type": "markdown", @@ -528,17 +526,15 @@ }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ "import time\n", "from IPython.display import clear_output\n", "\n", "# Poll job status every 10 seconds until completed\n", "while True:\n", - " status = client.customization.jobs.get_status(\n", - " name=job.name,\n", + " status = client.jobs.get_status(\n", + " name=job.job.name,\n", " workspace=\"default\"\n", " )\n", "\n", @@ -551,7 +547,7 @@ " training_phase: str | None = None\n", "\n", " for job_step in status.steps or []:\n", - " if job_step.name == \"customization-training-job\":\n", + " if job_step.name == \"training\":\n", " for task in job_step.tasks or []:\n", " task_details = task.status_details or {}\n", " step = task_details.get(\"step\")\n", @@ -568,13 +564,18 @@ " else:\n", " print(\"Training step not started yet or progress info not available\")\n", "\n", - " # Exit loop when job is completed (or failed/cancelled)\n", + " # Exit loop when job reaches a terminal status\n", " if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n", " print(f\"\\nJob finished with status: {status.status}\")\n", " break\n", "\n", - " time.sleep(10)" - ] + " time.sleep(10)\n", + "\n", + "if status.status != \"completed\":\n", + " raise RuntimeError(f\"Training job finished with status: {status.status}\")" + ], + "execution_count": null, + "outputs": [] }, { "cell_type": "markdown", @@ -607,23 +608,19 @@ }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ "# Validate model entity exists\n", - "model_entity = client.models.retrieve(workspace='default', name=job.spec.output.name)\n", + "model_entity = client.models.retrieve(workspace='default', name=OUTPUT_NAME)\n", "print(model_entity.model_dump_json(indent=2))" - ] + ], + "execution_count": null, + "outputs": [] }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ - "from nemo_platform.types.inference import NIMDeploymentParam\n", - "\n", "# Create deployment config\n", "deploy_suffix = uuid.uuid4().hex[:4]\n", "DEPLOYMENT_CONFIG_NAME = f\"sft-model-deployment-cfg-{deploy_suffix}\"\n", @@ -632,14 +629,16 @@ "deployment_config = client.inference.deployment_configs.create(\n", " workspace=\"default\",\n", " name=DEPLOYMENT_CONFIG_NAME,\n", - " nim_deployment=NIMDeploymentParam(\n", - " image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n", - " image_tag=\"1.15.5\",\n", - " gpu=1,\n", - " model_name=job.spec.output.name, # ModelEntity name from training,\n", - " model_namespace=\"default\", # Workspace where ModelEntity lives\n", - " additional_envs={\"NIM_MODEL_PROFILE\": \"vllm\"}\n", - " ),\n", + " engine=\"vllm\",\n", + " model_spec={\n", + " \"model_namespace\": \"default\",\n", + " \"model_name\": OUTPUT_NAME,\n", + " },\n", + " executor_config={\n", + " \"gpu\": 1,\n", + " \"image_name\": \"vllm/vllm-openai\",\n", + " \"image_tag\": \"v0.22.1\",\n", + " },\n", ")\n", "\n", "# Deploy model using deployment_config created above\n", @@ -657,8 +656,10 @@ ")\n", "\n", "print(f\"Deployment name: {deployment.name}\")\n", - "print(f\"Deployment status: {deployment_status.status}\")" - ] + "print(f\"Deployment status: {deployment_status.status}\")\n" + ], + "execution_count": null, + "outputs": [] }, { "cell_type": "markdown", @@ -667,76 +668,31 @@ "The deployment service automatically:\n", "- Downloads model weights from the Files service\n", "- Provisions storage (PVC) for the weights\n", - "- Configures and starts the NIM container\n", + "- Configures and starts the vLLM container\n", "\n", "**Multi-GPU Deployment:**\n", "\n", - "For larger models requiring multiple GPUs, configure parallelism with environment variables:\n", + "For larger models requiring multiple GPUs, increase `gpu` in `executor_config`. vLLM computes tensor parallelism from the GPU count and model architecture:\n", "\n", "```python\n", "deployment_config = client.inference.deployment_configs.create(\n", " workspace=\"default\",\n", " name=\"sft-model-config-multigpu\",\n", - " \n", - " nim_deployment={\n", - " \"image_name\": \"nvcr.io/nim/nvidia/llm-nim\",\n", - " \"image_tag\": \"1.13.1\",\n", - " \"gpu\": 2, # Total GPUs\n", - " \"additional_envs\": {\n", - " \"NIM_TENSOR_PARALLEL_SIZE\": \"2\", # Tensor parallelism\n", - " \"NIM_PIPELINE_PARALLEL_SIZE\": \"1\" # Pipeline parallelism\n", - " }\n", - " }\n", + " engine=\"vllm\",\n", + " model_spec={\n", + " \"model_namespace\": \"default\",\n", + " \"model_name\": OUTPUT_NAME,\n", + " },\n", + " executor_config={\n", + " \"gpu\": 2,\n", + " \"image_name\": \"vllm/vllm-openai\",\n", + " \"image_tag\": \"v0.22.1\",\n", + " },\n", ")\n", - "```" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "**Single-Node Constraint:** Model deployments are limited to a single node. The maximum `gpu` value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.\n", - "\n", - "---\n", - "\n", - "#### GPU Parallelism\n", - "\n", - "By default, NIM uses all GPUs for tensor parallelism (TP). You can customize this behavior using the `NIM_TENSOR_PARALLEL_SIZE` and `NIM_PIPELINE_PARALLEL_SIZE` environment variables.\n", - "\n", - "| Strategy | Description | Best For |\n", - "|----------|-------------|----------|\n", - "| **Tensor Parallel (TP)** | Splits model layers across GPUs | Lowest latency |\n", - "| **Pipeline Parallel (PP)** | Splits model depth across GPUs | Highest throughput |\n", - "\n", - "**Formula:** `gpu` = `NIM_TENSOR_PARALLEL_SIZE` × `NIM_PIPELINE_PARALLEL_SIZE`\n", - "\n", - "---\n", - "\n", - "#### Example Configurations\n", - "\n", - "**Default (TP=8, PP=1) — Lowest Latency**\n", - "```\n", - "\"gpu\": 8\n", - "# NIM automatically sets NIM_TENSOR_PARALLEL_SIZE=8\n", "```\n", "\n", - "**Balanced (TP=4, PP=2)**\n", - "```\n", - "\"gpu\": 8,\n", - "\"additional_envs\": {\n", - " \"NIM_TENSOR_PARALLEL_SIZE\": \"4\",\n", - " \"NIM_PIPELINE_PARALLEL_SIZE\": \"2\"\n", - "}\n", - "```\n", - "\n", - "**Throughput Optimized (TP=2, PP=4)**\n", - "```\n", - "\"gpu\": 8,\n", - "\"additional_envs\": {\n", - " \"NIM_TENSOR_PARALLEL_SIZE\": \"2\",\n", - " \"NIM_PIPELINE_PARALLEL_SIZE\": \"4\"\n", - "}\n", - "```" + "**Single-Node Constraint:** Model deployments are limited to a single node. The maximum `gpu` value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.\n", + "" ] }, { @@ -748,9 +704,7 @@ }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ "import time\n", "from IPython.display import clear_output\n", @@ -794,7 +748,9 @@ " raise TimeoutError(f\"Deployment timeout after {TIMEOUT_MINUTES} minutes\")\n", "\n", " time.sleep(15)" - ] + ], + "execution_count": null, + "outputs": [] }, { "cell_type": "markdown", @@ -809,9 +765,7 @@ }, { "cell_type": "code", - "execution_count": null, "metadata": {}, - "outputs": [], "source": [ "# Wait for deployment to be ready, then test\n", "# Test the fine-tuned model with a question answering prompt\n", @@ -827,7 +781,7 @@ " name=deployment.name,\n", " workspace=\"default\",\n", " body={\n", - " \"model\": f\"default/{job.spec.output.name}\",\n", + " \"model\": f\"default/{OUTPUT_NAME}\",\n", " \"messages\": messages,\n", " \"temperature\": 0,\n", " \"max_tokens\": 128\n", @@ -840,7 +794,9 @@ "print(f\"Question: {question}\")\n", "print(f\"Expected: Neil Armstrong\")\n", "print(f\"Model output: {response['choices'][0]['message']['content']}\")" - ] + ], + "execution_count": null, + "outputs": [] }, { "cell_type": "markdown", @@ -875,26 +831,26 @@ "\n", "**Job fails during model download:**\n", "- Verify authentication secrets are configured (refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n", - "- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n", - "- Check the `model_uri` format is correct (`fileset://`)\n", - "- Ensure you have accepted the model's terms of service on HuggingFace\n", - "- Check job status and logs: `client.customization.jobs.retrieve(name=job.name, workspace=\"default\")`\n", + "- For gated HuggingFace models (Llama, Gemma), accept the license on the model page (for example, [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct))\n", + "- Confirm the model fileset uses `token_secret=hf_secret.name` for gated models\n", + "- Check `AutomodelJobInput` references use the `workspace/name` format: `model=f\"default/{MODEL_NAME}\"` and `dataset={\"training\": f\"default/{DATASET_NAME}\"}` (for example, `default/llama-3-2-1b-base`, `default/sft-dataset`)\n", + "- Verify the model entity points at the fileset: `fileset=f\"default/{MODEL_NAME}\"`\n", + "- Check job status: `client.jobs.get_status(name=job.job.name, workspace=\"default\")`\n", "\n", "**Job fails with OOM (Out of Memory) error:**\n", - "1. **First try:** Reduce `micro_batch_size` from 2 to 1\n", - "2. **Still OOM:** Reduce `batch_size` from 4 to 2\n", - "3. **Still OOM:** Reduce `max_seq_length` from 2048 to 1024 or 512\n", - "4. **Last resort:** Increase GPU count and use `tensor_parallel_size` for model sharding\n", + "1. **First try:** Reduce `global_batch_size` from 64 to 32 or 16 in `batch={...}`\n", + "2. **Still OOM:** Keep `micro_batch_size` at 1 (already the minimum in this tutorial)\n", + "3. **Still OOM:** Reduce `max_seq_length` from 2048 to 1024 or 512 in `training={...}`\n", + "4. **Last resort:** Increase `num_gpus_per_node` and `tensor_parallel_size` in `parallelism={...}`\n", "\n", "**Loss curves not decreasing (underfitting):**\n", - "- Increase training duration: `epochs: 5-10` instead of 3\n", - "- Adjust learning rate: Try `1e-5` to `1e-4`\n", - "- Add warmup: Set `warmup_steps` to ~10% of total training steps\n", + "- Increase training duration: raise `epochs` from 2 to 3-5 in `schedule={...}`\n", + "- Adjust learning rate: try `1e-4` or `1e-5` instead of the default `5e-5` in `optimizer={...}`\n", "- Check data quality: Verify formatting, remove duplicates, ensure diversity\n", "\n", "**Training loss decreases but validation loss increases (overfitting):**\n", - "- Reduce epochs: Try `epochs: 1-2` instead of 5+\n", - "- Lower learning rate: Use `2e-5` or `1e-5`\n", + "- Reduce `epochs` from 2 to 1 in `schedule={...}`\n", + "- Lower `learning_rate` from `5e-5` to `2e-5` or `1e-5` in `optimizer={...}`\n", "- Increase dataset size and diversity\n", "- Verify train/validation split has no data leakage\n", "\n", @@ -902,14 +858,14 @@ "- Training metrics optimize for loss, not your actual task—evaluate on real use cases\n", "- Review data quality, format, and diversity—metrics can be misleading with poor data\n", "- Try a different base model size or architecture\n", - "- Adjust learning rate and batch size\n", + "- Adjust `learning_rate` and `global_batch_size`\n", "- Compare to baseline: Test base model to ensure fine-tuning improved performance\n", "\n", "**Deployment fails:**\n", - "- Verify output model exists: `client.models.retrieve(name=job.spec.output.name, workspace=\"default\")`\n", + "- Verify output model exists: `client.models.retrieve(name=OUTPUT_NAME, workspace=\"default\")`\n", "- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n", - "- Ensure sufficient GPU resources available for model size\n", - "- Verify NIM image tag `1.13.1` is compatible with your model\n", + "- Ensure sufficient GPU resources for `executor_config={\"gpu\": 1, ...}`\n", + "- Verify the deployment config matches this tutorial: `engine=\"vllm\"` with `vllm/vllm-openai:v0.22.1`\n", "\n", "\n", "## Next Steps\n", diff --git a/docs/customizer/tutorials/sft-customization-job.mdx b/docs/customizer/tutorials/sft-customization-job.mdx index 836bb4ce04..ea470754c7 100644 --- a/docs/customizer/tutorials/sft-customization-job.mdx +++ b/docs/customizer/tutorials/sft-customization-job.mdx @@ -3,7 +3,701 @@ title: "Full SFT Customization" description: "" --- - +[Run in Google Colab](https://colab.research.google.com/github/NVIDIA-NeMo/nemo-platform/blob/main/docs/customizer/tutorials/sft-customization-job.ipynb) + +# Full SFT Customization + +Learn how to fine-tune all model weights using supervised fine-tuning (SFT) to customize LLM behavior for your specific tasks. + +## About + +Supervised Fine-Tuning (SFT) customizes model behavior, injects new knowledge, and optimizes performance for specific domains and tasks. Full SFT modifies **all model weights** during training, providing maximum customization flexibility. + +**What you can achieve with SFT:** + +- 🎯 **Specialize for domains:** Fine-tune models on legal texts, medical records, or financial data +- 💡 **Inject knowledge:** Add new information not present in the base model +- 📈 **Improve accuracy:** Optimize for specific tasks like sentiment analysis, summarization, or code generation + +### SFT vs LoRA: Understanding the Trade-offs + +**Full SFT** trains all model parameters (for example, all 70 billion weights in Llama 70B): + +- ✅ Maximum model adaptation and knowledge injection +- ✅ Can fundamentally change model behavior +- ✅ Best for significant domain shifts or specialized tasks +- ❌ Requires substantial GPU resources (4-8x more than LoRA) +- ❌ Produces full model weights (~140GB for Llama 70B) +- ❌ Longer training time + +**LoRA** trains only ~1% of weights by adding thin matrices to existing weights: + +- ✅ 75-95% less memory required +- ✅ Faster training (2-4x speedup) +- ✅ Produces small adapter files (~100-500MB) +- ✅ Multiple adapters can share one base model +- ❌ Limited adaptation capability compared to full fine-tuning + +**When to choose Full SFT:** + +- Training small models (1B-8B) where resource cost is manageable +- Need fundamental behavior changes (for example, medical diagnosis, legal reasoning) +- Injecting substantial new knowledge not in the base model + +**When to choose LoRA:** Refer to the [LoRA tutorial](/documentation/customizer-reference/tutorials/lora-customization-job) for most use cases, especially with large models (70B+) or limited GPU resources. + +## Prerequisites + +Before starting this tutorial, ensure you have: + +1. **Completed the [Quickstart](/documentation/get-started)** to install and deploy NeMo Platform locally +2. **Installed the Python SDK** (PyPI wrapper: `pip install "nemo-platform[all]"`; source checkout: run `make bootstrap` from the repository root) + +## Quick Start + +### 1. Initialize SDK + +The SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](/documentation/get-started) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable: + +```sh +export NMP_BASE_URL= +``` + +```python +import json +import os +from nemo_platform import NeMoPlatform, ConflictError + +NMP_BASE_URL = os.environ.get("NMP_BASE_URL", "http://localhost:8080") +client = NeMoPlatform( + base_url=NMP_BASE_URL, + workspace="default" +) +``` + +### 2. Prepare Dataset + +Create your data in JSONL format—one JSON object per line. The platform auto-detects your data format. Supported dataset formats are listed below. + +**Flexible Data Setup:** +- **No validation file?** The platform automatically creates a 10% validation split +- **Multiple files?** Upload to `training/` or `validation/` subdirectories—they will be automatically merged +- **Format detection:** Your data format is auto-detected at training time + +In this tutorial the following dataset directory structure will be used: +``` +my_dataset +`-- training.jsonl +`-- validation.jsonl +``` + +#### Simple Prompt/Completion Format +The simplest format with input prompt and expected completion: +- **`prompt`**: The input prompt for the model +- **`completion`**: The expected output response + +```json +{"prompt": "Write an email to confirm our hotel reservation.", "completion": "Dear Hotel Team, I am writing to confirm our reservation for two guests..."} +``` + +#### Chat Format (for conversational models) +For multi-turn conversations, use the messages format: +- **`messages`**: List of message objects with `role` and `content` fields +- Roles: `system`, `user`, `assistant` + +```json +{"messages": [{"role": "system", "content": "You are a helpful assistant."}, {"role": "user", "content": "What is AI?"}, {"role": "assistant", "content": "AI is..."}]} +``` + +#### Custom Format (specify columns in job) +You can use custom field names and map them during job creation: +- Define your own field names +- Map them to prompt/completion in the job configuration + +```json +{"question": "What is 2+2?", "answer": "4"} +``` + +### 3. Create Dataset FileSet and Upload Training Data + +Install huggingface datasets package to download public [rajpurkar/squad](https://huggingface.co/datasets/rajpurkar/squad) dataset if it is not installed in your Python environment: + +```sh +pip install datasets +``` + +#### Download rajpurkar/squad Dataset + +SQuAD (Stanford Question Answering Dataset) is a reading comprehension dataset consisting of questions posed on Wikipedia articles, where the answer is a segment of text from the corresponding passage. + +```python +from pathlib import Path +from datasets import load_dataset, DatasetDict +import json + +# Load the SQuAD dataset from Hugging Face +print("Loading dataset rajpurkar/squad") +raw_dataset = load_dataset("rajpurkar/squad") +if not isinstance(raw_dataset, DatasetDict): + raise ValueError("Dataset does not contain expected splits") + +print("Loaded dataset") + +# Configuration +VALIDATION_PROPORTION = 0.05 +SEED = 1234 + +# For the purpose of this tutorial, we'll use a subset of the dataset +# The larger the datasets, the better the model will perform but longer the training will take +training_size = 3000 +validation_size = 300 +DATASET_PATH = Path("sft-dataset").absolute() + +# Create directory if it doesn't exist +os.makedirs(DATASET_PATH, exist_ok=True) + +# Get the train split and create a validation split from it +train_set = raw_dataset.get('train') +split_dataset = train_set.train_test_split(test_size=VALIDATION_PROPORTION, seed=SEED) + +# Select subsets for the tutorial +train_ds = split_dataset['train'].select(range(min(training_size, len(split_dataset['train'])))) +validation_ds = split_dataset['test'].select(range(min(validation_size, len(split_dataset['test'])))) + +# Convert SQuAD format to prompt/completion format and save to JSONL +def convert_squad_to_sft_format(example): + """Convert SQuAD format to prompt/completion format for SFT training.""" + prompt = f"Context: {example['context']} Question: {example['question']} Answer:" + completion = example["answers"]["text"][0] # Take the first answer + return {"prompt": prompt, "completion": completion} + +# Save training data +with open(f"{DATASET_PATH}/training.jsonl", "w", encoding="utf-8") as f: + for example in train_ds: + converted = convert_squad_to_sft_format(example) + f.write(json.dumps(converted) + "\n") + +# Save validation data +with open(f"{DATASET_PATH}/validation.jsonl", "w", encoding="utf-8") as f: + for example in validation_ds: + converted = convert_squad_to_sft_format(example) + f.write(json.dumps(converted) + "\n") + +print(f"Saved training.jsonl with {len(train_ds)} rows") +print(f"Saved validation.jsonl with {len(validation_ds)} rows") + +# Show a sample from the training data +print("\nSample from training data:") +with open(f"{DATASET_PATH}/training.jsonl", 'r') as f: + first_line = f.readline() + sample = json.loads(first_line) + print(f"Prompt: {sample['prompt'][:200]}...") + print(f"Completion: {sample['completion']}") +``` + +```python +# Create fileset to store SFT training data +DATASET_NAME = "sft-dataset" + +try: + client.files.filesets.create( + workspace="default", + name=DATASET_NAME, + description="SFT training data" + ) + print(f"Created fileset: {DATASET_NAME}") +except ConflictError: + print(f"Fileset '{DATASET_NAME}' already exists, continuing...") + +# Upload training data files individually to ensure correct structure +client.files.upload( + local_path=f"{DATASET_PATH}/", # Trailing slash uploads directory contents to fileset root + remote_path="", + fileset=DATASET_NAME, + workspace="default" +) + +# Validate training data is uploaded correctly +print("Training data:") +print(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2)) +``` + +### 4. Secrets Setup + +If you plan to use NGC or HuggingFace models, you will need to configure authentication: + +- **NGC models** (`ngc://` URIs): Requires NGC API key +- **HuggingFace models** (`hf://` URIs): Requires HF token for gated/private models + + +Configure these as secrets in your platform. Refer to [Managing Secrets](/documentation/get-started/core-concepts/manage-secrets) for detailed instructions. + +Get your credentials to access base models: +- [NGC API Key](https://ngc.nvidia.com/) (Setup → Generate API Key) +- [HuggingFace Token](https://huggingface.co/settings/tokens) (Create token with Read access) + + +--- + +#### Quick Setup Example + +In this tutorial we are going to work with [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model from HuggingFace. Ensure that you have sufficient permissions to download the model. If you cannot access the files on the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) Hugging Face page, request access + +**HuggingFace Authentication:** +- For gated models (Llama, Gemma), you must provide a HuggingFace token via the `token_secret` parameter +- Get your token from [HuggingFace Settings](https://huggingface.co/settings/tokens) (requires Read access) +- Accept the model's terms on the HuggingFace model page before using it. Example: [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) +- For public models, you can omit the `token_secret` parameter when creating a fileset for model in the next step + +```python +# Export the HF_TOKEN and NGC_API_KEY environment variables if they are not already set +HF_TOKEN = os.getenv("HF_TOKEN") +NGC_API_KEY = os.getenv("NGC_API_KEY") + + +def create_or_get_secret(name: str, value: str | None, label: str): + if not value: + raise ValueError(f"{label} is not set") + try: + secret = client.secrets.create( + name=name, + workspace="default", + value=value, + ) + print(f"Created secret: {name}") + return secret + except ConflictError: + print(f"Secret '{name}' already exists, continuing...") + return client.secrets.retrieve(name=name, workspace="default") + + +# Create HuggingFace token secret +hf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN") +print("HF_TOKEN secret:") +print(hf_secret.model_dump_json(indent=2)) + +# Create NGC API key secret +# Uncomment the line below if you have NGC API Key and want to finetune NGC models +# ngc_api_key = create_or_get_secret("ngc-api-key", NGC_API_KEY, "NGC_API_KEY") +``` + +### 5. Create Base Model FileSet and Model Entity + +Create a fileset pointing to [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model in HuggingFace that we will train with SFT. Then create a Model Entity that references this fileset. Model downloading will take place at training time. + +Note: for public models, you can omit the `token_secret` parameter when creating a model fileset. + +```python +import time + +# Create a fileset pointing to the desired HuggingFace model +from nemo_platform.types.files import HuggingfaceStorageConfigParam + +HF_REPO_ID = "meta-llama/Llama-3.2-1B-Instruct" +MODEL_NAME = "llama-3-2-1b-base" + +# Ensure you have a HuggingFace token secret created +try: + base_model_fs = client.files.filesets.create( + workspace="default", + name=MODEL_NAME, + description="Llama 3.2 1B base model from HuggingFace", + storage=HuggingfaceStorageConfigParam( + type="huggingface", + # repo_id is the full model name from Hugging Face + repo_id=HF_REPO_ID, + repo_type="model", + # we use the secret created in the previous step + token_secret=hf_secret.name + ) + ) + print(f"Created base model fileset: {MODEL_NAME}") +except ConflictError: + print(f"Base model fileset already exists. Skipping creation.") + base_model_fs = client.files.filesets.retrieve( + workspace="default", + name=MODEL_NAME, + ) + +# Create the Model Entity representation. +try: + base_model = client.models.create( + workspace="default", + name=MODEL_NAME, + fileset=f"default/{MODEL_NAME}", + ) + print(f"Created Model Entity: {MODEL_NAME}") +except ConflictError: + print(f"Base model already exists. Updating fileset if different.") + base_model = client.models.update( + workspace="default", + name=MODEL_NAME, + fileset=f"default/{MODEL_NAME}", + ) + +print(f"\nBase model fileset: fileset://default/{base_model.name}") +print("Base model fileset files list:") +print(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace="default").data], indent=2)) + +# Wait for ModelSpec to be populated from the checkpoint +print("\nWaiting for ModelSpec to be populated...") +SPEC_TIMEOUT_SECONDS = 120 +spec_start = time.time() +while not base_model.spec: + if time.time() - spec_start > SPEC_TIMEOUT_SECONDS: + raise TimeoutError(f"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds") + time.sleep(2) + base_model = client.models.retrieve( + workspace="default", + name=MODEL_NAME, + ) + +print(f"ModelSpec populated: {base_model.spec}") +``` + +### 6. Create SFT Finetuning Job +Create a customization job to fine-tune all model weights using the **Automodel** backend and `AutomodelJobInput`. + +**GPU Requirements:** +- 1B models: 1 GPU (24GB+ VRAM) +- 3B models: 1-2 GPUs +- 8B models: 2-4 GPUs +- 70B models: 8+ GPUs + +Adjust `num_gpus_per_node` based on your model size. + +```python +import uuid +from nemo_automodel_plugin.schema import AutomodelJobInput + +job_suffix = uuid.uuid4().hex[:4] + +JOB_NAME = f"my-sft-job-{job_suffix}" +OUTPUT_NAME = f"sft-model-{job_suffix}" + +spec = AutomodelJobInput( + model=f"default/{base_model.name}", + dataset={"training": f"default/{DATASET_NAME}"}, + training={ + "training_type": "sft", + "finetuning_type": "all_weights", + "max_seq_length": 2048, + }, + schedule={"epochs": 2}, + batch={"global_batch_size": 64, "micro_batch_size": 1}, + optimizer={"learning_rate": 5e-5}, + parallelism={ + "num_gpus_per_node": 1, + "num_nodes": 1, + "tensor_parallel_size": 1, + "pipeline_parallel_size": 1, + }, + output={"name": OUTPUT_NAME}, +) + +job = client.customization.automodel.jobs.create( + spec=spec, workspace="default", name=JOB_NAME +) + +print(f"Submitted job: {job.job.name}") +print(f"Output model: {OUTPUT_NAME}") + +``` + +### 7. Track Training Progress + +```python +import time +from IPython.display import clear_output + +# Poll job status every 10 seconds until completed +while True: + status = client.jobs.get_status( + name=job.job.name, + workspace="default" + ) + + clear_output(wait=True) + print(f"Job Status: {status.model_dump_json(indent=2)}") + + # Extract training progress from nested steps structure + step: int | None = None + max_steps: int | None = None + training_phase: str | None = None + + for job_step in status.steps or []: + if job_step.name == "training": + for task in job_step.tasks or []: + task_details = task.status_details or {} + step = task_details.get("step") + max_steps = task_details.get("max_steps") + training_phase = task_details.get("phase") + break + break + + if step is not None and max_steps is not None: + progress_pct = (step / max_steps) * 100 + print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)") + if training_phase: + print(f"Training Phase: {training_phase}") + else: + print("Training step not started yet or progress info not available") + + # Exit loop when job reaches a terminal status + if status.status in ("completed", "failed", "cancelled", "error"): + print(f"\nJob finished with status: {status.status}") + break + + time.sleep(10) + +if status.status != "completed": + raise RuntimeError(f"Training job finished with status: {status.status}") +``` + +**Interpreting SFT Training Metrics:** + +Monitor the relationship between training and validation loss curves: + +| Scenario | Interpretation | Action | +|----------|----------------|--------| +| **Both decreasing together** | Model is learning well | Continue training | +| **Training decreases, validation flat/increasing** | Overfitting | Reduce epochs, add data | +| **Both flat/not decreasing** | Underfitting | Increase LR, check data | +| **Sudden spikes** | Training instability | Lower learning rate | + +**Note:** Training metrics measure optimization progress, not final model quality. Always evaluate the deployed model on your specific use case. + +(ft-deploy-full-weight-model)= + +### 8. Deploy Fine-Tuned Model + +After training completes, deploy using the Deployment Management Service: + +```python +# Validate model entity exists +model_entity = client.models.retrieve(workspace='default', name=OUTPUT_NAME) +print(model_entity.model_dump_json(indent=2)) +``` + +```python +# Create deployment config +deploy_suffix = uuid.uuid4().hex[:4] +DEPLOYMENT_CONFIG_NAME = f"sft-model-deployment-cfg-{deploy_suffix}" +DEPLOYMENT_NAME = f"sft-model-deployment-{deploy_suffix}" + +deployment_config = client.inference.deployment_configs.create( + workspace="default", + name=DEPLOYMENT_CONFIG_NAME, + engine="vllm", + model_spec={ + "model_namespace": "default", + "model_name": OUTPUT_NAME, + }, + executor_config={ + "gpu": 1, + "image_name": "vllm/vllm-openai", + "image_tag": "v0.22.1", + }, +) + +# Deploy model using deployment_config created above +deployment = client.inference.deployments.create( + workspace="default", + name=DEPLOYMENT_NAME, + config=deployment_config.name +) + + +# Check deployment status +deployment_status = client.inference.deployments.retrieve( + name=deployment.name, + workspace="default" +) + +print(f"Deployment name: {deployment.name}") +print(f"Deployment status: {deployment_status.status}") + +``` + +The deployment service automatically: +- Downloads model weights from the Files service +- Provisions storage (PVC) for the weights +- Configures and starts the vLLM container + +**Multi-GPU Deployment:** + +For larger models requiring multiple GPUs, increase `gpu` in `executor_config`. vLLM computes tensor parallelism from the GPU count and model architecture: + +```python +deployment_config = client.inference.deployment_configs.create( + workspace="default", + name="sft-model-config-multigpu", + engine="vllm", + model_spec={ + "model_namespace": "default", + "model_name": OUTPUT_NAME, + }, + executor_config={ + "gpu": 2, + "image_name": "vllm/vllm-openai", + "image_tag": "v0.22.1", + }, +) +``` + +**Single-Node Constraint:** Model deployments are limited to a single node. The maximum `gpu` value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported. + +### Track Deployment Status + +```python +import time +from IPython.display import clear_output + +# Poll deployment status every 15 seconds until ready +TIMEOUT_MINUTES = 30 +start_time = time.time() +timeout_seconds = TIMEOUT_MINUTES * 60 + +print(f"Monitoring deployment '{deployment.name}'...") +print(f"Timeout: {TIMEOUT_MINUTES} minutes\n") + +while True: + deployment_status = client.inference.deployments.retrieve( + name=deployment.name, + workspace="default" + ) + + elapsed = time.time() - start_time + elapsed_min = int(elapsed // 60) + elapsed_sec = int(elapsed % 60) + + clear_output(wait=True) + print(f"Deployment: {deployment.name}") + print(f"Status: {deployment_status.status}") + print(f"Elapsed time: {elapsed_min}m {elapsed_sec}s") + + # Check if deployment is ready + if deployment_status.status == "READY": + print("\nDeployment is ready!") + if not client.models.wait_for_gateway(deployment.name, workspace="default", timeout=60): + raise RuntimeError("Inference gateway did not become ready") + break + + # Check for failure states + if deployment_status.status in ("FAILED", "ERROR", "TERMINATED", "LOST"): + raise RuntimeError(f"Deployment failed with status: {deployment_status.status}") + + # Check timeout + if elapsed > timeout_seconds: + raise TimeoutError(f"Deployment timeout after {TIMEOUT_MINUTES} minutes") + + time.sleep(15) +``` + +### 9. Evaluate Your Model + +After training, evaluate whether your model meets your requirements: + +#### Quick Manual Evaluation + +```python +# Wait for deployment to be ready, then test +# Test the fine-tuned model with a question answering prompt +context = "The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit." +question = "Who was the first person to walk on the Moon?" + +messages = [ + {"role": "user", "content": f"Based on the following context, answer the question.\n\nContext: {context}\n\nQuestion: {question}"} +] + +response = client.inference.gateway.provider.post( + "v1/chat/completions", + name=deployment.name, + workspace="default", + body={ + "model": f"default/{OUTPUT_NAME}", + "messages": messages, + "temperature": 0, + "max_tokens": 128 + } +) + +print("=" * 60) +print("MODEL EVALUATION") +print("=" * 60) +print(f"Question: {question}") +print(f"Expected: Neil Armstrong") +print(f"Model output: {response['choices'][0]['message']['content']}") +``` + +#### Evaluation Best Practices + +**Manual Evaluation** (Recommended) +- Test with real-world examples from your use case +- Compare responses to base model and expected outputs +- Verify the model exhibits desired behavior changes +- Check edge cases and error handling + +**What to look for:** +- ✅ Model follows your desired output format +- ✅ Applies domain knowledge correctly +- ✅ Maintains general language capabilities +- ✅ Avoids unwanted behaviors or biases +- ❌ Doesn't hallucinate facts not in training data +- ❌ Doesn't produce repetitive or nonsensical outputs + +--- + +## Hyperparameters + +For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](/documentation/customizer-reference/manage-customization-jobs/training-configuration). + +--- + + +## Troubleshooting + +**Job fails during model download:** +- Verify authentication secrets are configured (refer to [Managing Secrets](/documentation/get-started/core-concepts/manage-secrets)) +- For gated HuggingFace models (Llama, Gemma), accept the license on the model page (for example, [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct)) +- Confirm the model fileset uses `token_secret=hf_secret.name` for gated models +- Check `AutomodelJobInput` references use the `workspace/name` format: `model=f"default/{MODEL_NAME}"` and `dataset={"training": f"default/{DATASET_NAME}"}` (for example, `default/llama-3-2-1b-base`, `default/sft-dataset`) +- Verify the model entity points at the fileset: `fileset=f"default/{MODEL_NAME}"` +- Check job status: `client.jobs.get_status(name=job.job.name, workspace="default")` + +**Job fails with OOM (Out of Memory) error:** +1. **First try:** Reduce `global_batch_size` from 64 to 32 or 16 in `batch={...}` +2. **Still OOM:** Keep `micro_batch_size` at 1 (already the minimum in this tutorial) +3. **Still OOM:** Reduce `max_seq_length` from 2048 to 1024 or 512 in `training={...}` +4. **Last resort:** Increase `num_gpus_per_node` and `tensor_parallel_size` in `parallelism={...}` + +**Loss curves not decreasing (underfitting):** +- Increase training duration: raise `epochs` from 2 to 3-5 in `schedule={...}` +- Adjust learning rate: try `1e-4` or `1e-5` instead of the default `5e-5` in `optimizer={...}` +- Check data quality: Verify formatting, remove duplicates, ensure diversity + +**Training loss decreases but validation loss increases (overfitting):** +- Reduce `epochs` from 2 to 1 in `schedule={...}` +- Lower `learning_rate` from `5e-5` to `2e-5` or `1e-5` in `optimizer={...}` +- Increase dataset size and diversity +- Verify train/validation split has no data leakage + +**Model output quality is poor despite good training metrics:** +- Training metrics optimize for loss, not your actual task—evaluate on real use cases +- Review data quality, format, and diversity—metrics can be misleading with poor data +- Try a different base model size or architecture +- Adjust `learning_rate` and `global_batch_size` +- Compare to baseline: Test base model to ensure fine-tuning improved performance + +**Deployment fails:** +- Verify output model exists: `client.models.retrieve(name=OUTPUT_NAME, workspace="default")` +- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace="default")` +- Ensure sufficient GPU resources for `executor_config={"gpu": 1, ...}` +- Verify the deployment config matches this tutorial: `engine="vllm"` with `vllm/vllm-openai:v0.22.1` + + +## Next Steps + +- [Monitor training metrics](/documentation/customizer-reference/tutorials/metrics) in detail +- [Evaluate your fine-tuned model](/documentation/evaluate-models) using the Evaluator service +- Learn about [LoRA customization](/documentation/customizer-reference/tutorials/lora-customization-job) for resource-efficient fine-tuning diff --git a/docs/customizer/tutorials/understand-configurations-and-models.mdx b/docs/customizer/tutorials/understand-configurations-and-models.mdx index 4385fdf8ce..5eb1b09d5b 100644 --- a/docs/customizer/tutorials/understand-configurations-and-models.mdx +++ b/docs/customizer/tutorials/understand-configurations-and-models.mdx @@ -120,20 +120,20 @@ print(f"Parameters: {model.spec.base_num_parameters:,}") **3. Create a Customization Job** ```python -job = client.customization.jobs.create( +from nemo_automodel_plugin.schema import AutomodelJobInput + +spec = AutomodelJobInput( + model="default/llama-3-2-1b", + dataset={"training": "default/email-training-data"}, + training={"training_type": "sft", "finetuning_type": "lora", "lora": {"rank": 8, "alpha": 32}}, + schedule={"epochs": 3}, + batch={"global_batch_size": 32, "micro_batch_size": 1}, +) + +job = client.customization.automodel.jobs.create( workspace="default", name="my-email-assistant-lora", - spec={ - "model": "default/llama-3-2-1b", - "dataset": "fileset://default/email-training-data", - "training": { - "type": "sft", - "peft": {"type": "lora", "rank": 8, "alpha": 32}, - "epochs": 3, - "batch_size": 32, - }, - "deployment_config": {"lora_enabled": True}, - }, + spec=spec, ) ``` @@ -446,15 +446,19 @@ models = client.models.list(workspace="default") # Get a specific Model Entity with adapters model = client.models.retrieve(workspace="default", name="llama-3-2-1b") -# Create a customization job -job = client.customization.jobs.create( +# Create an Automodel training job +from nemo_automodel_plugin.schema import AutomodelJobInput + +spec = AutomodelJobInput( + model="default/llama-3-2-1b", + dataset={"training": "default/my-dataset"}, + training={"training_type": "sft", "finetuning_type": "lora"}, +) + +job = client.customization.automodel.jobs.create( workspace="default", name="my-job", - spec={ - "model": "default/llama-3-2-1b", - "dataset": "fileset://default/my-dataset", - "training": {"type": "sft", "peft": {"type": "lora"}}, - }, + spec=spec, ) # Add an adapter to a model diff --git a/docs/example-applications/tool-calling.ipynb b/docs/example-applications/tool-calling.ipynb index b337237a23..295f2b20b9 100644 --- a/docs/example-applications/tool-calling.ipynb +++ b/docs/example-applications/tool-calling.ipynb @@ -192,7 +192,7 @@ " prompt=(\n", " \"Generate 2-4 realistic API function definitions for the '{% raw %}{{ domain }}{% endraw %}' domain. \"\n", " \"Each function should have a descriptive snake_case name, clear description, \"\n", - " \"and well-typed parameters. Make them diverse — include functions with \"\n", + " \"and well-typed parameters. Make them diverse \u2014 include functions with \"\n", " \"different parameter counts and types.\"\n", " ),\n", " output_format=ToolDefinitions,\n", @@ -222,7 +222,7 @@ " \"Given this user query: '{% raw %}{{ user_query }}{% endraw %}'\\n\"\n", " \"And these available tools: {% raw %}{{ tools }}{% endraw %}\\n\\n\"\n", " \"Determine exactly which tool should be called and with what arguments. \"\n", - " \"Be precise with argument values — they should directly address the user's request.\"\n", + " \"Be precise with argument values \u2014 they should directly address the user's request.\"\n", " ),\n", " output_format=ExpectedToolCalls,\n", " model_alias=MODEL_ALIAS,\n", @@ -298,7 +298,7 @@ "\n", "The preview functionality lets you iterate quickly on your data to ensure the right structure and quality. Once ready, the `create` method is used to generate the full training dataset.\n", "\n", - "For this tutorial we generate **60 samples** (~5 minutes) to keep iteration fast. For production use, generate **500+ samples** (~1 hour) for significantly better results — in our testing, 500 samples yielded **~93% function name accuracy**, compared to ~1.4% for the un-fine-tuned base model.\n" + "For this tutorial we generate **60 samples** (~5 minutes) to keep iteration fast. For production use, generate **500+ samples** (~1 hour) for significantly better results \u2014 in our testing, 500 samples yielded **~93% function name accuracy**, compared to ~1.4% for the un-fine-tuned base model.\n" ] }, { @@ -340,7 +340,7 @@ "### Convert Dataset\n", "\n", "- Fine-tuning requires the [OpenAI messages+tools format](https://platform.openai.com/docs/guides/function-calling) with specific nesting.\n", - "- Data Designer generated flat structures — now we wrap them in the OpenAI format.\n", + "- Data Designer generated flat structures \u2014 now we wrap them in the OpenAI format.\n", "- Key transformations: wrap tools/calls in `{\"type\": \"function\", \"function\": {...}}`, convert parameter lists to properties dicts, add `\"content\": \"\"` to the assistant message." ] }, @@ -397,7 +397,7 @@ "\n", "training_data = dataset.apply(to_openai_format, axis=1, result_type=\"expand\")\n", "\n", - "# Llama 3.2 1B's chat template supports only single tool calls per turn — filter out any\n", + "# Llama 3.2 1B's chat template supports only single tool calls per turn \u2014 filter out any\n", "# generated samples that contain multiple tool calls in the assistant response.\n", "pre_filter = len(training_data)\n", "training_data = training_data[\n", @@ -559,7 +559,7 @@ "print(f\"xLAM dataset size: {len(xlam_dataset)}\")\n", "\n", "xlam_converted = [r for ex in xlam_dataset if (r := convert_xlam_example(ex)) is not None]\n", - "print(f\"After filtering (single tool-call, ≤{LIMIT_TOOL_PROPERTIES} params): {len(xlam_converted)}\")\n", + "print(f\"After filtering (single tool-call, \u2264{LIMIT_TOOL_PROPERTIES} params): {len(xlam_converted)}\")\n", "\n", "eval_size = max(1, int(0.15 * len(train_data)))\n", "eval_rows = random.sample(xlam_converted, min(eval_size, len(xlam_converted)))\n", @@ -596,6 +596,7 @@ "\n", "try:\n", " client.files.filesets.create(\n", + " workspace=WORKSPACE,\n", " name=DATASET_NAME,\n", " description=\"synthetic tool-calling training and validation data in OpenAI chat format\",\n", " )\n", @@ -607,13 +608,14 @@ " local_path=DATASET_PATH,\n", " remote_path=\"\",\n", " fileset=DATASET_NAME,\n", + " workspace=WORKSPACE,\n", ")\n", "\n", "print(\"\\nTraining data files:\")\n", "print(json.dumps(\n", - " [f.model_dump() for f in client.files.list(fileset=DATASET_NAME).data],\n", + " [f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=WORKSPACE).data],\n", " indent=2,\n", - "))" + "))\n" ] }, { @@ -635,6 +637,7 @@ "source": [ "try:\n", " client.files.filesets.create(\n", + " workspace=WORKSPACE,\n", " name=EVAL_DATASET_NAME,\n", " description=\"synthetic tool-calling evaluation data (messages, tools, ground-truth tool_calls)\",\n", " )\n", @@ -646,13 +649,14 @@ " local_path=EVAL_DATASET_PATH,\n", " remote_path=\"\",\n", " fileset=EVAL_DATASET_NAME,\n", + " workspace=WORKSPACE,\n", ")\n", "\n", "print(\"\\nEvaluation data files:\")\n", "print(json.dumps(\n", - " [f.model_dump() for f in client.files.list(fileset=EVAL_DATASET_NAME).data],\n", + " [f.model_dump() for f in client.files.list(fileset=EVAL_DATASET_NAME, workspace=WORKSPACE).data],\n", " indent=2,\n", - "))" + "))\n" ] }, { @@ -676,7 +680,31 @@ "id": "0bd91637", "metadata": {}, "outputs": [], - "source": "HF_TOKEN = os.environ.get(\"HF_TOKEN\")\nif HF_TOKEN is None:\n raise ValueError(\"HF_TOKEN is not set.\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n name=name,\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name)\n\n\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(hf_secret.model_dump_json(indent=2))" + "source": [ + "HF_TOKEN = os.environ.get(\"HF_TOKEN\")\n", + "if HF_TOKEN is None:\n", + " raise ValueError(\"HF_TOKEN is not set.\")\n", + "\n", + "\n", + "def create_or_get_secret(name: str, value: str | None, label: str):\n", + " if not value:\n", + " raise ValueError(f\"{label} is not set\")\n", + " try:\n", + " secret = client.secrets.create(\n", + " workspace=WORKSPACE,\n", + " name=name,\n", + " value=value,\n", + " )\n", + " print(f\"Created secret: {name}\")\n", + " return secret\n", + " except ConflictError:\n", + " print(f\"Secret '{name}' already exists, continuing...\")\n", + " return client.secrets.retrieve(name=name, workspace=WORKSPACE)\n", + "\n", + "\n", + "hf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\n", + "print(hf_secret.model_dump_json(indent=2))\n" + ] }, { "cell_type": "markdown", @@ -708,6 +736,7 @@ "\n", "try:\n", " base_model_fs = client.files.filesets.create(\n", + " workspace=WORKSPACE,\n", " name=MODEL_NAME,\n", " description=\"Llama 3.2 1B Instruct base model from HuggingFace\",\n", " storage=HuggingfaceStorageConfigParam(\n", @@ -729,11 +758,13 @@ "except ConflictError:\n", " print(f\"Base model fileset already exists. Skipping creation.\")\n", " base_model_fs = client.files.filesets.retrieve(\n", + " workspace=WORKSPACE,\n", " name=MODEL_NAME,\n", " )\n", "\n", "try:\n", " base_model = client.models.create(\n", + " workspace=WORKSPACE,\n", " name=MODEL_NAME,\n", " fileset=f\"{WORKSPACE}/{MODEL_NAME}\",\n", " )\n", @@ -741,6 +772,7 @@ "except ConflictError:\n", " print(f\"Base model already exists. Updating fileset if different.\")\n", " base_model = client.models.update(\n", + " workspace=WORKSPACE,\n", " name=MODEL_NAME,\n", " fileset=f\"{WORKSPACE}/{MODEL_NAME}\",\n", " )\n", @@ -748,7 +780,7 @@ "print(f\"\\nBase model fileset: fileset://{WORKSPACE}/{base_model.name}\")\n", "print(\"Base model fileset files list:\")\n", "print(json.dumps(\n", - " [f.model_dump() for f in client.files.list(fileset=MODEL_NAME).data],\n", + " [f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=WORKSPACE).data],\n", " indent=2,\n", "))\n", "\n", @@ -761,10 +793,11 @@ " raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n", " time.sleep(2)\n", " base_model = client.models.retrieve(\n", + " workspace=WORKSPACE,\n", " name=MODEL_NAME,\n", " )\n", "\n", - "print(f\"ModelSpec populated: {base_model.spec.model_dump()}\")" + "print(f\"ModelSpec populated: {base_model.spec.model_dump()}\")\n" ] }, { @@ -774,12 +807,7 @@ "source": [ "## 4. Create LoRA Fine-Tuning Job\n", "\n", - "Create a customization job using SFT training with LoRA PEFT. The `peft=LoRaParamsParam()` parameter enables LoRA instead of full-weight fine-tuning.\n", - "\n", - "**LoRA defaults** (can be overridden in `LoRaParamsParam()`):\n", - "- `rank`: LoRA rank (dimensionality of the low-rank matrices)\n", - "- `alpha`: Scaling factor for LoRA updates\n", - "- `target_modules`: Which model layers to apply LoRA to" + "Create an **Automodel** customization job using `AutomodelJobInput` with `finetuning_type: lora`. After training completes, deploy the base model with LoRA support in the next section.\n" ] }, { @@ -790,45 +818,33 @@ "outputs": [], "source": [ "import uuid\n", - "from nemo_platform.types.customization import (\n", - " CustomizationJobInputParam,\n", - " SftTrainingParam,\n", - " ParallelismParamsParam,\n", - " LoRaParamsParam,\n", - " DeploymentParamsParam,\n", - ")\n", + "from nemo_automodel_plugin.schema import AutomodelJobInput\n", "\n", "job_suffix = uuid.uuid4().hex[:4]\n", "JOB_NAME = f\"tool-calling-lora-{job_suffix}\"\n", + "OUTPUT_NAME = f\"tool-calling-adapter-{job_suffix}\"\n", + "\n", + "spec = AutomodelJobInput(\n", + " model=f\"{WORKSPACE}/{base_model.name}\",\n", + " dataset={\"training\": f\"{WORKSPACE}/{DATASET_NAME}\"},\n", + " training={\n", + " \"training_type\": \"sft\",\n", + " \"finetuning_type\": \"lora\",\n", + " \"max_seq_length\": 2048,\n", + " },\n", + " schedule={\"epochs\": 4},\n", + " batch={\"global_batch_size\": 4, \"micro_batch_size\": 1},\n", + " optimizer={\"learning_rate\": 1e-4},\n", + " parallelism={\"num_gpus_per_node\": 1},\n", + " output={\"name\": OUTPUT_NAME},\n", + ")\n", "\n", - "\n", - "job = client.customization.jobs.create(\n", - " name=JOB_NAME,\n", - " spec=CustomizationJobInputParam(\n", - " model=f\"{WORKSPACE}/{base_model.name}\",\n", - " dataset=f\"fileset://{WORKSPACE}/{DATASET_NAME}\",\n", - " training=SftTrainingParam(\n", - " type=\"sft\",\n", - " epochs=4,\n", - " batch_size=4,\n", - " learning_rate=0.0001,\n", - " max_seq_length=2048,\n", - " micro_batch_size=1,\n", - " peft=LoRaParamsParam(),\n", - " parallelism=ParallelismParamsParam(\n", - " num_gpus_per_node=1,\n", - " num_nodes=1,\n", - " tensor_parallel_size=1,\n", - " pipeline_parallel_size=1,\n", - " ),\n", - " ),\n", - " deployment_config=DeploymentParamsParam(\n", - " lora_enabled=True,\n", - " ),\n", - " ),\n", + "job = client.customization.automodel.jobs.create(\n", + " spec=spec, workspace=WORKSPACE, name=JOB_NAME\n", ")\n", "\n", - "print(job.model_dump_json(indent=2))" + "print(f\"Submitted job: {job.job.name}\")\n", + "print(f\"Output adapter: {OUTPUT_NAME}\")\n" ] }, { @@ -849,7 +865,7 @@ "import time\n", "from IPython.display import clear_output\n", "\n", - "TERMINAL_JOB_STATUSES = {\"completed\", \"cancelled\", \"error\"}\n", + "TERMINAL_JOB_STATUSES = {\"completed\", \"failed\", \"cancelled\", \"error\"}\n", "\n", "\n", "def wait_for_job(poll_fn, label, timeout_minutes=60, poll_interval=10, display_fn=None):\n", @@ -899,7 +915,7 @@ "def training_progress(status):\n", " \"\"\"Extract and display training step progress.\"\"\"\n", " for step in status.steps or []:\n", - " if step.name == \"customization-training-job\":\n", + " if step.name == \"training\":\n", " for task in step.tasks or []:\n", " details = task.status_details or {}\n", " s, mx = details.get(\"step\"), details.get(\"max_steps\")\n", @@ -910,7 +926,7 @@ " return\n", " print(\"Training step not started yet\")\n", "\n", - "print(\"Defined wait_for_job helper function\")" + "print(\"Defined wait_for_job helper function\")\n" ] }, { @@ -921,11 +937,14 @@ "outputs": [], "source": [ "job_status = wait_for_job(\n", - " poll_fn=lambda: client.customization.jobs.get_status(name=job.name),\n", + " poll_fn=lambda: client.jobs.get_status(name=job.job.name, workspace=WORKSPACE),\n", " label=\"Training\",\n", " timeout_minutes=120,\n", " display_fn=training_progress,\n", - ")" + ")\n", + "\n", + "if job_status.status != \"completed\":\n", + " raise RuntimeError(f\"Training job finished with status: {job_status.status}\")\n" ] }, { @@ -933,9 +952,9 @@ "id": "6d4eb20b", "metadata": {}, "source": [ - "## 5. Verify Auto-Deployed Model\n", + "## 5. Deploy Fine-Tuned Model\n", "\n", - "Since we set `lora_enabled=True` in the customization job's `deployment_config`, the platform automatically creates a NIM deployment for the base model after training completes. The LoRA adapter is attached to the base model entity (enabled by default) and the deployment serves both the base weights and the adapter through a single NIM instance.\n" + "After training completes, verify the LoRA adapter is attached to the base model entity, then create a NIM deployment with `lora_enabled=True` so both the base weights and adapter are served through a single deployment.\n" ] }, { @@ -945,16 +964,16 @@ "metadata": {}, "outputs": [], "source": [ - "ADAPTER_NAME = job.spec.output.name\n", + "ADAPTER_NAME = OUTPUT_NAME\n", "print(f\"Looking for adapter: {ADAPTER_NAME}\")\n", "\n", "# The adapter may not be attached to the model entity immediately after\n", - "# training completes — poll until it appears.\n", + "# training completes \u2014 poll until it appears.\n", "ADAPTER_TIMEOUT = 120\n", "adapter_start = time.time()\n", "adapter = None\n", "while time.time() - adapter_start < ADAPTER_TIMEOUT:\n", - " base_model = client.models.retrieve(name=MODEL_NAME)\n", + " base_model = client.models.retrieve(name=MODEL_NAME, workspace=WORKSPACE)\n", " matches = [a for a in (base_model.adapters or []) if a.name == ADAPTER_NAME]\n", " if matches:\n", " adapter = matches[0]\n", @@ -968,7 +987,36 @@ " )\n", "\n", "print(f\"Base model: {base_model.name}\")\n", - "print(f\"Adapter:\\n{adapter.model_dump_json(indent=2)}\")" + "print(f\"Adapter:\\n{adapter.model_dump_json(indent=2)}\")\n", + "\n", + "deploy_suffix = uuid.uuid4().hex[:4]\n", + "DEPLOYMENT_CONFIG_NAME = f\"tool-calling-deploy-cfg-{deploy_suffix}\"\n", + "deployment_name = f\"tool-calling-deploy-{deploy_suffix}\"\n", + "\n", + "deployment_config = client.inference.deployment_configs.create(\n", + " workspace=WORKSPACE,\n", + " name=DEPLOYMENT_CONFIG_NAME,\n", + " engine=\"vllm\",\n", + " model_spec={\n", + " \"model_namespace\": WORKSPACE,\n", + " \"model_name\": MODEL_NAME,\n", + " \"lora_enabled\": True,\n", + " },\n", + " executor_config={\n", + " \"gpu\": 1,\n", + " \"image_name\": \"vllm/vllm-openai\",\n", + " \"image_tag\": \"v0.22.1\",\n", + " \"additional_args\": [\"--max-lora-rank\", \"32\"],\n", + " },\n", + ")\n", + "\n", + "deployment = client.inference.deployments.create(\n", + " workspace=WORKSPACE,\n", + " name=deployment_name,\n", + " config=deployment_config.name,\n", + ")\n", + "\n", + "print(f\"Deployment status: {deployment.status}\")\n" ] }, { @@ -986,18 +1034,17 @@ "metadata": {}, "outputs": [], "source": [ - "DEPLOYMENT_NAME = f\"sft-deploy-{MODEL_NAME}\"\n", - "\n", "TIMEOUT_MINUTES = 30\n", "start_time = time.time()\n", "timeout_seconds = TIMEOUT_MINUTES * 60\n", "\n", - "print(f\"Monitoring deployment '{DEPLOYMENT_NAME}'...\")\n", + "print(f\"Monitoring deployment '{deployment_name}'...\")\n", "print(f\"Timeout: {TIMEOUT_MINUTES} minutes\\n\")\n", "\n", "while True:\n", " deployment_status = client.inference.deployments.retrieve(\n", - " name=DEPLOYMENT_NAME,\n", + " name=deployment_name,\n", + " workspace=WORKSPACE,\n", " )\n", "\n", " elapsed = time.time() - start_time\n", @@ -1005,13 +1052,13 @@ " elapsed_sec = int(elapsed % 60)\n", "\n", " clear_output(wait=True)\n", - " print(f\"Deployment: {DEPLOYMENT_NAME}\")\n", + " print(f\"Deployment: {deployment_name}\")\n", " print(f\"Status: {deployment_status.status}\")\n", " print(f\"Elapsed time: {elapsed_min}m {elapsed_sec}s\")\n", "\n", - " if deployment_status.status == \"READY\":\n", + " if deployment_status.status in (\"RUNNING\", \"READY\"):\n", " print(\"\\nDeployment is ready!\")\n", - " if not client.models.wait_for_gateway(DEPLOYMENT_NAME, workspace=WORKSPACE, timeout=60):\n", + " if not client.models.wait_for_gateway(deployment_name, workspace=WORKSPACE, timeout=60):\n", " raise RuntimeError(\"Inference gateway did not become ready\")\n", " break\n", "\n", @@ -1021,7 +1068,7 @@ " if elapsed > timeout_seconds:\n", " raise TimeoutError(f\"Deployment timeout after {TIMEOUT_MINUTES} minutes\")\n", "\n", - " time.sleep(15)" + " time.sleep(15)\n" ] }, { @@ -1068,10 +1115,12 @@ "\n", "def test_tool_calling(model_name: str, label: str):\n", " \"\"\"Send a tool calling request and display the response.\"\"\"\n", - " response = client.inference.gateway.model.post(\n", + " response = client.inference.gateway.provider.post(\n", " \"v1/chat/completions\",\n", - " name=model_name,\n", + " name=deployment_name,\n", + " workspace=WORKSPACE,\n", " body={\n", + " \"model\": model_name,\n", " \"messages\": test_messages,\n", " \"tools\": test_tools,\n", " \"tool_choice\": \"auto\",\n", @@ -1087,7 +1136,7 @@ "\n", "\n", "test_tool_calling(MODEL_NAME, \"BASE MODEL (before fine-tuning)\")\n", - "test_tool_calling(ADAPTER_NAME, \"FINE-TUNED MODEL (after LoRA)\")" + "test_tool_calling(ADAPTER_NAME, \"FINE-TUNED MODEL (after LoRA)\")\n" ] }, { diff --git a/docs/fern/components/notebooks/distillation-customization-job.json b/docs/fern/components/notebooks/distillation-customization-job.json index c072c5276f..969cdfa323 100644 --- a/docs/fern/components/notebooks/distillation-customization-job.json +++ b/docs/fern/components/notebooks/distillation-customization-job.json @@ -7,8 +7,8 @@ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)\n3. **Installed evaluation dependencies:**\n\n```sh\npip install evaluate rouge_score datasets\n```", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
  5. Installed evaluation dependencies:
  6. \n
\n
pip install evaluate rouge_score datasets\n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n3. **Installed evaluation dependencies:**\n\n```sh\npip install evaluate rouge_score datasets\n```", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (PyPI wrapper: pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  4. \n
  5. Installed evaluation dependencies:
  6. \n
\n
pip install evaluate rouge_score datasets\n
\n" }, { "type": "markdown", @@ -34,9 +34,9 @@ }, { "type": "code", - "source": "DATASET_NAME = \"kd-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"Knowledge distillation training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\nprint(\"Uploaded files:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", + "source": "DATASET_NAME = \"kd-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"Knowledge distillation training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=f\"{DATASET_PATH}/\",\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\nprint(\"Uploaded files:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", "language": "python", - "source_html": "DATASET_NAME = "kd-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="Knowledge distillation training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\nprint("Uploaded files:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" + "source_html": "DATASET_NAME = "kd-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="Knowledge distillation training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=f"{DATASET_PATH}/",\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\nprint("Uploaded files:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" }, { "type": "markdown", @@ -67,15 +67,15 @@ }, { "type": "code", - "source": "from nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n DistillationTrainingParam,\n ParallelismParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\n\nTEACHER_JOB_NAME = f\"teacher-sft-job-{job_suffix}\"\n\nteacher_job = client.customization.jobs.create(\n name=TEACHER_JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{teacher_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nTRAINED_TEACHER_NAME = teacher_job.spec.output.name\nprint(f\"Teacher training job: {teacher_job.name}\")\nprint(f\"Output teacher model: {TRAINED_TEACHER_NAME}\")", + "source": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\n\nTEACHER_JOB_NAME = f\"teacher-sft-job-{job_suffix}\"\nTEACHER_OUTPUT_NAME = f\"teacher-model-{job_suffix}\"\n\nteacher_spec = AutomodelJobInput(\n model=f\"default/{teacher_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"all_weights\",\n \"max_seq_length\": 2048,\n },\n schedule={\"epochs\": 1},\n batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": TEACHER_OUTPUT_NAME},\n)\n\nteacher_job = client.customization.automodel.jobs.create(\n spec=teacher_spec, workspace=\"default\", name=TEACHER_JOB_NAME\n)\n\nTRAINED_TEACHER_NAME = TEACHER_OUTPUT_NAME\nprint(f\"Teacher training job: {teacher_job.job.name}\")\nprint(f\"Output teacher model: {TRAINED_TEACHER_NAME}\")", "language": "python", - "source_html": "from nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n DistillationTrainingParam,\n ParallelismParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\n\nTEACHER_JOB_NAME = f"teacher-sft-job-{job_suffix}"\n\nteacher_job = client.customization.jobs.create(\n name=TEACHER_JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{teacher_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nTRAINED_TEACHER_NAME = teacher_job.spec.output.name\nprint(f"Teacher training job: {teacher_job.name}")\nprint(f"Output teacher model: {TRAINED_TEACHER_NAME}")\n" + "source_html": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\n\nTEACHER_JOB_NAME = f"teacher-sft-job-{job_suffix}"\nTEACHER_OUTPUT_NAME = f"teacher-model-{job_suffix}"\n\nteacher_spec = AutomodelJobInput(\n model=f"default/{teacher_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "all_weights",\n "max_seq_length": 2048,\n },\n schedule={"epochs": 1},\n batch={"global_batch_size": 64, "micro_batch_size": 1},\n optimizer={"learning_rate": 5e-5},\n parallelism={"num_gpus_per_node": 1},\n output={"name": TEACHER_OUTPUT_NAME},\n)\n\nteacher_job = client.customization.automodel.jobs.create(\n spec=teacher_spec, workspace="default", name=TEACHER_JOB_NAME\n)\n\nTRAINED_TEACHER_NAME = TEACHER_OUTPUT_NAME\nprint(f"Teacher training job: {teacher_job.job.name}")\nprint(f"Output teacher model: {TRAINED_TEACHER_NAME}")\n" }, { "type": "code", - "source": "from IPython.display import clear_output\n\n\ndef wait_for_job(job_name: str):\n \"\"\"Poll job status until completion.\"\"\"\n while True:\n status = client.customization.jobs.get_status(name=job_name, workspace=\"default\")\n clear_output(wait=True)\n print(f\"Job: {job_name}\")\n print(f\"Status: {status.status}\")\n\n for job_step in status.steps or []:\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n details = task.status_details or {}\n step = details.get(\"step\")\n max_steps = details.get(\"max_steps\")\n if step is not None and max_steps is not None:\n print(f\"Progress: Step {step}/{max_steps} ({step / max_steps * 100:.1f}%)\")\n phase = details.get(\"phase\")\n if phase:\n print(f\"Phase: {phase}\")\n break\n break\n\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished: {status.status}\")\n return status\n\n time.sleep(10)\n\n\nteacher_status = wait_for_job(TEACHER_JOB_NAME)", + "source": "from IPython.display import clear_output\n\n\ndef wait_for_job(job_name: str):\n \"\"\"Poll job status until completion.\"\"\"\n while True:\n status = client.jobs.get_status(name=job_name, workspace=\"default\")\n clear_output(wait=True)\n print(f\"Job: {job_name}\")\n print(f\"Status: {status.status}\")\n\n for job_step in status.steps or []:\n if job_step.name == \"training\":\n for task in job_step.tasks or []:\n details = task.status_details or {}\n step = details.get(\"step\")\n max_steps = details.get(\"max_steps\")\n if step is not None and max_steps is not None:\n print(f\"Progress: Step {step}/{max_steps} ({step / max_steps * 100:.1f}%)\")\n phase = details.get(\"phase\")\n if phase:\n print(f\"Phase: {phase}\")\n break\n break\n\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished: {status.status}\")\n return status\n\n time.sleep(10)\n\n\nteacher_status = wait_for_job(TEACHER_JOB_NAME)\nassert teacher_status.status == \"completed\"", "language": "python", - "source_html": "from IPython.display import clear_output\n\n\ndef wait_for_job(job_name: str):\n """Poll job status until completion."""\n while True:\n status = client.customization.jobs.get_status(name=job_name, workspace="default")\n clear_output(wait=True)\n print(f"Job: {job_name}")\n print(f"Status: {status.status}")\n\n for job_step in status.steps or []:\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n details = task.status_details or {}\n step = details.get("step")\n max_steps = details.get("max_steps")\n if step is not None and max_steps is not None:\n print(f"Progress: Step {step}/{max_steps} ({step / max_steps * 100:.1f}%)")\n phase = details.get("phase")\n if phase:\n print(f"Phase: {phase}")\n break\n break\n\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished: {status.status}")\n return status\n\n time.sleep(10)\n\n\nteacher_status = wait_for_job(TEACHER_JOB_NAME)\n" + "source_html": "from IPython.display import clear_output\n\n\ndef wait_for_job(job_name: str):\n """Poll job status until completion."""\n while True:\n status = client.jobs.get_status(name=job_name, workspace="default")\n clear_output(wait=True)\n print(f"Job: {job_name}")\n print(f"Status: {status.status}")\n\n for job_step in status.steps or []:\n if job_step.name == "training":\n for task in job_step.tasks or []:\n details = task.status_details or {}\n step = details.get("step")\n max_steps = details.get("max_steps")\n if step is not None and max_steps is not None:\n print(f"Progress: Step {step}/{max_steps} ({step / max_steps * 100:.1f}%)")\n phase = details.get("phase")\n if phase:\n print(f"Phase: {phase}")\n break\n break\n\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished: {status.status}")\n return status\n\n time.sleep(10)\n\n\nteacher_status = wait_for_job(TEACHER_JOB_NAME)\nassert teacher_status.status == "completed"\n" }, { "type": "markdown", @@ -84,15 +84,15 @@ }, { "type": "code", - "source": "from nemo_platform.types.inference import NIMDeploymentParam\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f\"baseline-student-cfg-{baseline_suffix}\"\nBASELINE_DEPLOYMENT_NAME = f\"baseline-student-{baseline_suffix}\"\n\nbaseline_deployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n image_tag=\"1.15.5\",\n gpu=1,\n model_name=student_model.name,\n model_namespace=\"default\",\n additional_envs={\"NIM_MODEL_PROFILE\": \"vllm\"}\n ),\n)\n\nbaseline_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_deployment_config.name\n)\n\nprint(f\"Baseline student deployment: {baseline_deployment.name}\")", + "source": "baseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f\"baseline-student-cfg-{baseline_suffix}\"\nBASELINE_DEPLOYMENT_NAME = f\"baseline-student-{baseline_suffix}\"\n\nbaseline_deployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_CONFIG,\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": student_model.name,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n },\n)\n\nbaseline_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_deployment_config.name\n)\n\nprint(f\"Baseline student deployment: {baseline_deployment.name}\")", "language": "python", - "source_html": "from nemo_platform.types.inference import NIMDeploymentParam\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f"baseline-student-cfg-{baseline_suffix}"\nBASELINE_DEPLOYMENT_NAME = f"baseline-student-{baseline_suffix}"\n\nbaseline_deployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name="nvcr.io/nim/nvidia/llm-nim",\n image_tag="1.15.5",\n gpu=1,\n model_name=student_model.name,\n model_namespace="default",\n additional_envs={"NIM_MODEL_PROFILE": "vllm"}\n ),\n)\n\nbaseline_deployment = client.inference.deployments.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_deployment_config.name\n)\n\nprint(f"Baseline student deployment: {baseline_deployment.name}")\n" + "source_html": "baseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f"baseline-student-cfg-{baseline_suffix}"\nBASELINE_DEPLOYMENT_NAME = f"baseline-student-{baseline_suffix}"\n\nbaseline_deployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_CONFIG,\n engine="vllm",\n model_spec={\n "model_namespace": "default",\n "model_name": student_model.name,\n },\n executor_config={\n "gpu": 1,\n "image_name": "vllm/vllm-openai",\n "image_tag": "v0.22.1",\n },\n)\n\nbaseline_deployment = client.inference.deployments.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_deployment_config.name\n)\n\nprint(f"Baseline student deployment: {baseline_deployment.name}")\n" }, { "type": "code", - "source": "def wait_for_deployment(deployment_name: str, timeout_minutes: int = 30):\n \"\"\"Poll deployment until ready.\"\"\"\n start = time.time()\n timeout = timeout_minutes * 60\n while True:\n dep = client.inference.deployments.retrieve(name=deployment_name, workspace=\"default\")\n elapsed = time.time() - start\n clear_output(wait=True)\n print(f\"Deployment: {deployment_name}\")\n print(f\"Status: {dep.status}\")\n print(f\"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s\")\n\n if dep.status == \"READY\":\n print(\"\\nDeployment is ready!\")\n return dep\n if dep.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n print(f\"\\nDeployment failed: {dep.status}\")\n return dep\n if elapsed > timeout:\n print(f\"\\nTimeout ({timeout_minutes}m). Check status manually.\")\n return dep\n time.sleep(15)\n\n\nwait_for_deployment(BASELINE_DEPLOYMENT_NAME)", + "source": "def wait_for_deployment(deployment_name: str, timeout_minutes: int = 30):\n \"\"\"Poll deployment until ready.\"\"\"\n start = time.time()\n timeout = timeout_minutes * 60\n while True:\n dep = client.inference.deployments.retrieve(name=deployment_name, workspace=\"default\")\n elapsed = time.time() - start\n clear_output(wait=True)\n print(f\"Deployment: {deployment_name}\")\n print(f\"Status: {dep.status}\")\n print(f\"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s\")\n\n if dep.status == \"READY\":\n print(\"\\nDeployment is ready!\")\n return dep\n if dep.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n print(f\"\\nDeployment failed: {dep.status}\")\n return dep\n if elapsed > timeout:\n print(f\"\\nTimeout ({timeout_minutes}m). Check status manually.\")\n return dep\n time.sleep(15)\n\n\ndep_status = wait_for_deployment(BASELINE_DEPLOYMENT_NAME)\nassert dep_status.status == \"READY\"", "language": "python", - "source_html": "def wait_for_deployment(deployment_name: str, timeout_minutes: int = 30):\n """Poll deployment until ready."""\n start = time.time()\n timeout = timeout_minutes * 60\n while True:\n dep = client.inference.deployments.retrieve(name=deployment_name, workspace="default")\n elapsed = time.time() - start\n clear_output(wait=True)\n print(f"Deployment: {deployment_name}")\n print(f"Status: {dep.status}")\n print(f"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s")\n\n if dep.status == "READY":\n print("\\nDeployment is ready!")\n return dep\n if dep.status in ("FAILED", "ERROR", "TERMINATED", "LOST"):\n print(f"\\nDeployment failed: {dep.status}")\n return dep\n if elapsed > timeout:\n print(f"\\nTimeout ({timeout_minutes}m). Check status manually.")\n return dep\n time.sleep(15)\n\n\nwait_for_deployment(BASELINE_DEPLOYMENT_NAME)\n" + "source_html": "def wait_for_deployment(deployment_name: str, timeout_minutes: int = 30):\n """Poll deployment until ready."""\n start = time.time()\n timeout = timeout_minutes * 60\n while True:\n dep = client.inference.deployments.retrieve(name=deployment_name, workspace="default")\n elapsed = time.time() - start\n clear_output(wait=True)\n print(f"Deployment: {deployment_name}")\n print(f"Status: {dep.status}")\n print(f"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s")\n\n if dep.status == "READY":\n print("\\nDeployment is ready!")\n return dep\n if dep.status in ("FAILED", "ERROR", "TERMINATED", "LOST"):\n print(f"\\nDeployment failed: {dep.status}")\n return dep\n if elapsed > timeout:\n print(f"\\nTimeout ({timeout_minutes}m). Check status manually.")\n return dep\n time.sleep(15)\n\n\ndep_status = wait_for_deployment(BASELINE_DEPLOYMENT_NAME)\nassert dep_status.status == "READY"\n" }, { "type": "markdown", @@ -118,9 +118,9 @@ }, { "type": "code", - "source": "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace=\"default\")\nprint(f\"Deleted baseline deployment: {BASELINE_DEPLOYMENT_NAME}\")\n\n# wait for deployment to be deleted\ntime.sleep(60)\n\nclient.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace=\"default\")\nprint(f\"Deleted baseline deployment config: {BASELINE_DEPLOYMENT_CONFIG}\")", + "source": "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace=\"default\")\nprint(f\"Deleted baseline deployment: {BASELINE_DEPLOYMENT_NAME}\")\n\nif not client.models.wait_for_status(\n deployment_name=BASELINE_DEPLOYMENT_NAME,\n desired_status=\"DELETED\",\n workspace=\"default\",\n timeout=600,\n):\n raise TimeoutError(\n f\"Deployment {BASELINE_DEPLOYMENT_NAME} was not deleted within timeout\"\n )\n\nclient.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace=\"default\")\nprint(f\"Deleted baseline deployment config: {BASELINE_DEPLOYMENT_CONFIG}\")", "language": "python", - "source_html": "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace="default")\nprint(f"Deleted baseline deployment: {BASELINE_DEPLOYMENT_NAME}")\n\n# wait for deployment to be deleted\ntime.sleep(60)\n\nclient.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace="default")\nprint(f"Deleted baseline deployment config: {BASELINE_DEPLOYMENT_CONFIG}")\n" + "source_html": "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace="default")\nprint(f"Deleted baseline deployment: {BASELINE_DEPLOYMENT_NAME}")\n\nif not client.models.wait_for_status(\n deployment_name=BASELINE_DEPLOYMENT_NAME,\n desired_status="DELETED",\n workspace="default",\n timeout=600,\n):\n raise TimeoutError(\n f"Deployment {BASELINE_DEPLOYMENT_NAME} was not deleted within timeout"\n )\n\nclient.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace="default")\nprint(f"Deleted baseline deployment config: {BASELINE_DEPLOYMENT_CONFIG}")\n" }, { "type": "markdown", @@ -134,9 +134,9 @@ }, { "type": "code", - "source": "KD_JOB_NAME = f\"my-kd-job-{job_suffix}\"\n\nkd_job = client.customization.jobs.create(\n name=KD_JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{student_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=DistillationTrainingParam(\n type=\"distillation\",\n teacher_model=f\"default/{TRAINED_TEACHER_NAME}\",\n teacher_precision=\"bf16\",\n distillation_ratio=0.5,\n distillation_temperature=2.0,\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nDISTILLED_STUDENT_NAME = kd_job.spec.output.name\nprint(f\"Distillation job: {kd_job.name}\")\nprint(f\"Output student model: {DISTILLED_STUDENT_NAME}\")", + "source": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\nKD_JOB_NAME = f\"my-kd-job-{job_suffix}\"\nKD_OUTPUT_NAME = f\"kd-student-{job_suffix}\"\n\nkd_spec = AutomodelJobInput(\n model=f\"default/{student_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"distillation\",\n \"finetuning_type\": \"all_weights\",\n \"teacher_model\": f\"default/{TRAINED_TEACHER_NAME}\",\n \"teacher_precision\": \"bf16\",\n \"distillation_ratio\": 0.5,\n \"distillation_temperature\": 2.0,\n \"max_seq_length\": 2048,\n },\n schedule={\"epochs\": 1},\n batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": KD_OUTPUT_NAME},\n)\n\nkd_job = client.customization.automodel.jobs.create(\n spec=kd_spec, workspace=\"default\", name=KD_JOB_NAME\n)\n\nDISTILLED_STUDENT_NAME = KD_OUTPUT_NAME\nprint(f\"Distillation job: {kd_job.job.name}\")\nprint(f\"Output student model: {DISTILLED_STUDENT_NAME}\")", "language": "python", - "source_html": "KD_JOB_NAME = f"my-kd-job-{job_suffix}"\n\nkd_job = client.customization.jobs.create(\n name=KD_JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{student_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=DistillationTrainingParam(\n type="distillation",\n teacher_model=f"default/{TRAINED_TEACHER_NAME}",\n teacher_precision="bf16",\n distillation_ratio=0.5,\n distillation_temperature=2.0,\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nDISTILLED_STUDENT_NAME = kd_job.spec.output.name\nprint(f"Distillation job: {kd_job.name}")\nprint(f"Output student model: {DISTILLED_STUDENT_NAME}")\n" + "source_html": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\nKD_JOB_NAME = f"my-kd-job-{job_suffix}"\nKD_OUTPUT_NAME = f"kd-student-{job_suffix}"\n\nkd_spec = AutomodelJobInput(\n model=f"default/{student_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "distillation",\n "finetuning_type": "all_weights",\n "teacher_model": f"default/{TRAINED_TEACHER_NAME}",\n "teacher_precision": "bf16",\n "distillation_ratio": 0.5,\n "distillation_temperature": 2.0,\n "max_seq_length": 2048,\n },\n schedule={"epochs": 1},\n batch={"global_batch_size": 64, "micro_batch_size": 1},\n optimizer={"learning_rate": 5e-5},\n parallelism={"num_gpus_per_node": 1},\n output={"name": KD_OUTPUT_NAME},\n)\n\nkd_job = client.customization.automodel.jobs.create(\n spec=kd_spec, workspace="default", name=KD_JOB_NAME\n)\n\nDISTILLED_STUDENT_NAME = KD_OUTPUT_NAME\nprint(f"Distillation job: {kd_job.job.name}")\nprint(f"Output student model: {DISTILLED_STUDENT_NAME}")\n" }, { "type": "markdown", @@ -145,9 +145,9 @@ }, { "type": "code", - "source": "kd_status = wait_for_job(KD_JOB_NAME)", + "source": "kd_status = wait_for_job(KD_JOB_NAME)\nassert kd_status.status == \"completed\"", "language": "python", - "source_html": "kd_status = wait_for_job(KD_JOB_NAME)\n" + "source_html": "kd_status = wait_for_job(KD_JOB_NAME)\nassert kd_status.status == "completed"\n" }, { "type": "markdown", @@ -156,9 +156,9 @@ }, { "type": "code", - "source": "deploy_suffix_2 = uuid.uuid4().hex[:4]\nSTUDENT_DEPLOYMENT_CONFIG = f\"kd-student-deploy-cfg-{deploy_suffix_2}\"\nSTUDENT_DEPLOYMENT_NAME = f\"kd-student-deploy-{deploy_suffix_2}\"\n\nstudent_deployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=STUDENT_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n image_tag=\"1.15.5\",\n gpu=1,\n model_name=DISTILLED_STUDENT_NAME,\n model_namespace=\"default\",\n additional_envs={\"NIM_MODEL_PROFILE\": \"vllm\"},\n ),\n)\n\nstudent_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=STUDENT_DEPLOYMENT_NAME,\n config=student_deployment_config.name\n)\n\nprint(f\"Student deployment: {student_deployment.name}\")", + "source": "deploy_suffix_2 = uuid.uuid4().hex[:4]\nSTUDENT_DEPLOYMENT_CONFIG = f\"kd-student-deploy-cfg-{deploy_suffix_2}\"\nSTUDENT_DEPLOYMENT_NAME = f\"kd-student-deploy-{deploy_suffix_2}\"\n\nstudent_deployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=STUDENT_DEPLOYMENT_CONFIG,\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": DISTILLED_STUDENT_NAME,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n },\n)\n\nstudent_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=STUDENT_DEPLOYMENT_NAME,\n config=student_deployment_config.name\n)\n\nprint(f\"Student deployment: {student_deployment.name}\")", "language": "python", - "source_html": "deploy_suffix_2 = uuid.uuid4().hex[:4]\nSTUDENT_DEPLOYMENT_CONFIG = f"kd-student-deploy-cfg-{deploy_suffix_2}"\nSTUDENT_DEPLOYMENT_NAME = f"kd-student-deploy-{deploy_suffix_2}"\n\nstudent_deployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=STUDENT_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name="nvcr.io/nim/nvidia/llm-nim",\n image_tag="1.15.5",\n gpu=1,\n model_name=DISTILLED_STUDENT_NAME,\n model_namespace="default",\n additional_envs={"NIM_MODEL_PROFILE": "vllm"},\n ),\n)\n\nstudent_deployment = client.inference.deployments.create(\n workspace="default",\n name=STUDENT_DEPLOYMENT_NAME,\n config=student_deployment_config.name\n)\n\nprint(f"Student deployment: {student_deployment.name}")\n" + "source_html": "deploy_suffix_2 = uuid.uuid4().hex[:4]\nSTUDENT_DEPLOYMENT_CONFIG = f"kd-student-deploy-cfg-{deploy_suffix_2}"\nSTUDENT_DEPLOYMENT_NAME = f"kd-student-deploy-{deploy_suffix_2}"\n\nstudent_deployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=STUDENT_DEPLOYMENT_CONFIG,\n engine="vllm",\n model_spec={\n "model_namespace": "default",\n "model_name": DISTILLED_STUDENT_NAME,\n },\n executor_config={\n "gpu": 1,\n "image_name": "vllm/vllm-openai",\n "image_tag": "v0.22.1",\n },\n)\n\nstudent_deployment = client.inference.deployments.create(\n workspace="default",\n name=STUDENT_DEPLOYMENT_NAME,\n config=student_deployment_config.name\n)\n\nprint(f"Student deployment: {student_deployment.name}")\n" }, { "type": "code", @@ -196,8 +196,8 @@ }, { "type": "markdown", - "source": "**Interpreting ROUGE Scores:**\n\n| Metric | Measures |\n|--------|----------|\n| **ROUGE-1** | Unigram overlap between prediction and reference |\n| **ROUGE-2** | Bigram overlap (captures phrase-level similarity) |\n| **ROUGE-L** | Longest common subsequence (captures sentence structure) |\n| **ROUGE-Lsum** | ROUGE-L computed over full summaries |\n\n**What to expect:**\n- The base student (1B, no training) provides a lower bound since it has not seen the task data\n- The distilled student (1B, KD) should significantly outperform the base student, demonstrating the knowledge transferred from the 3B teacher\n- If the distilled student scores are not much higher than the baseline, try increasing `distillation_temperature`, adjusting `distillation_ratio`, or training for more epochs\n\n---\n\n## Hyperparameters\n\nFor detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n\n---\n\n## Troubleshooting\n\n**Job fails during model download:**\n- Verify authentication secrets are configured (refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n- Check both `model` (student) and `teacher_model` URNs are correct\n- Ensure both model entities exist: `client.models.retrieve(name=..., workspace=\"default\")`\n\n**Job fails with OOM (Out of Memory) error:**\n\nKD loads both models, so OOM is more likely than with SFT:\n1. **First try:** Use `teacher_precision=\"bf16\"` to reduce teacher memory\n2. **Still OOM:** Reduce `micro_batch_size` to 1\n3. **Still OOM:** Reduce `batch_size` and `max_seq_length`\n4. **Last resort:** Increase `num_gpus_per_node`\n\n**No chat template / `/chat/completions` fails:**\n- Use Instruct model variants (e.g., `Llama-3.2-1B-Instruct`) instead of base models (`Llama-3.2-1B`). Base models do not include a chat template in their tokenizer, so the output model will also lack one.\n\n**Distilled model quality is poor:**\n- Increase `distillation_temperature` (try 2.0–5.0) to transfer more nuanced knowledge\n- Adjust `distillation_ratio`—if dataset labels are high-quality, lower the ratio; if the teacher is strong, raise it\n- Increase `epochs` or `max_steps` for more training\n- Verify teacher and student share the same vocabulary\n\n**Vocabulary mismatch error:**\n- Teacher and student must use the same tokenizer. Use models from the same family (e.g., Llama 3.2 1B Instruct + Llama 3.2 3B Instruct)\n\n**Deployment fails:**\n- Verify output model exists: `client.models.retrieve(name=job.spec.output.name, workspace=\"default\")`\n- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n- The distilled model has the same size as the student, so GPU requirements match the student model\n\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning\n- Learn about [Full SFT](./sft-customization-job) for direct supervised fine-tuning", - "source_html": "

Interpreting ROUGE Scores:

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
MetricMeasures
ROUGE-1Unigram overlap between prediction and reference
ROUGE-2Bigram overlap (captures phrase-level similarity)
ROUGE-LLongest common subsequence (captures sentence structure)
ROUGE-LsumROUGE-L computed over full summaries
\n

What to expect:

\n
    \n
  • The base student (1B, no training) provides a lower bound since it has not seen the task data
  • \n
  • The distilled student (1B, KD) should significantly outperform the base student, demonstrating the knowledge transferred from the 3B teacher
  • \n
  • If the distilled student scores are not much higher than the baseline, try increasing distillation_temperature, adjusting distillation_ratio, or training for more epochs
  • \n
\n
\n

Hyperparameters

\n

For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the Hyperparameter Reference.

\n
\n

Troubleshooting

\n

Job fails during model download:

\n
    \n
  • Verify authentication secrets are configured (refer to Managing Secrets)
  • \n
  • For gated HuggingFace models (Llama, Gemma), accept the license on the model page
  • \n
  • Check both model (student) and teacher_model URNs are correct
  • \n
  • Ensure both model entities exist: client.models.retrieve(name=..., workspace="default")
  • \n
\n

Job fails with OOM (Out of Memory) error:

\n

KD loads both models, so OOM is more likely than with SFT:

\n
    \n
  1. First try: Use teacher_precision="bf16" to reduce teacher memory
  2. \n
  3. Still OOM: Reduce micro_batch_size to 1
  4. \n
  5. Still OOM: Reduce batch_size and max_seq_length
  6. \n
  7. Last resort: Increase num_gpus_per_node
  8. \n
\n

No chat template / /chat/completions fails:

\n
    \n
  • Use Instruct model variants (e.g., Llama-3.2-1B-Instruct) instead of base models (Llama-3.2-1B). Base models do not include a chat template in their tokenizer, so the output model will also lack one.
  • \n
\n

Distilled model quality is poor:

\n
    \n
  • Increase distillation_temperature (try 2.0–5.0) to transfer more nuanced knowledge
  • \n
  • Adjust distillation_ratio—if dataset labels are high-quality, lower the ratio; if the teacher is strong, raise it
  • \n
  • Increase epochs or max_steps for more training
  • \n
  • Verify teacher and student share the same vocabulary
  • \n
\n

Vocabulary mismatch error:

\n
    \n
  • Teacher and student must use the same tokenizer. Use models from the same family (e.g., Llama 3.2 1B Instruct + Llama 3.2 3B Instruct)
  • \n
\n

Deployment fails:

\n
    \n
  • Verify output model exists: client.models.retrieve(name=job.spec.output.name, workspace="default")
  • \n
  • Check deployment logs: client.inference.deployments.get_logs(name=deployment.name, workspace="default")
  • \n
  • The distilled model has the same size as the student, so GPU requirements match the student model
  • \n
\n

Next Steps

\n\n" + "source": "**Interpreting ROUGE Scores:**\n\n| Metric | Measures |\n|--------|----------|\n| **ROUGE-1** | Unigram overlap between prediction and reference |\n| **ROUGE-2** | Bigram overlap (captures phrase-level similarity) |\n| **ROUGE-L** | Longest common subsequence (captures sentence structure) |\n| **ROUGE-Lsum** | ROUGE-L computed over full summaries |\n\n**What to expect:**\n- The base student (1B, no training) provides a lower bound since it has not seen the task data\n- The distilled student (1B, KD) should significantly outperform the base student, demonstrating the knowledge transferred from the 3B teacher\n- If the distilled student scores are not much higher than the baseline, try increasing `distillation_temperature`, adjusting `distillation_ratio`, or training for more epochs\n\n---\n\n## Hyperparameters\n\nFor detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n\n---\n\n## Troubleshooting\n\n**Job fails during model download:**\n- Verify authentication secrets are configured (refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n- Check both `model` (student) and `teacher_model` URNs are correct\n- Ensure both model entities exist: `client.models.retrieve(name=..., workspace=\"default\")`\n\n**Job fails with OOM (Out of Memory) error:**\n\nKD loads both models, so OOM is more likely than with SFT:\n1. **First try:** Use `teacher_precision=\"bf16\"` to reduce teacher memory\n2. **Still OOM:** Reduce `micro_batch_size` to 1\n3. **Still OOM:** Reduce `global_batch_size` and `max_seq_length`\n4. **Last resort:** Increase `num_gpus_per_node`\n\n**No chat template / `/chat/completions` fails:**\n- Use Instruct model variants (e.g., `Llama-3.2-1B-Instruct`) instead of base models (`Llama-3.2-1B`). Base models do not include a chat template in their tokenizer, so the output model will also lack one.\n\n**Distilled model quality is poor:**\n- Increase `distillation_temperature` (try 2.0–5.0) to transfer more nuanced knowledge\n- Adjust `distillation_ratio`—if dataset labels are high-quality, lower the ratio; if the teacher is strong, raise it\n- Increase `epochs` or `max_steps` for more training\n- Verify teacher and student share the same vocabulary\n\n**Vocabulary mismatch error:**\n- Teacher and student must use the same tokenizer. Use models from the same family (e.g., Llama 3.2 1B Instruct + Llama 3.2 3B Instruct)\n\n**Deployment fails:**\n- Verify output model exists: `client.models.retrieve(name=DISTILLED_STUDENT_NAME, workspace=\"default\")`\n- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n- The distilled model has the same size as the student, so GPU requirements match the student model\n\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning\n- Learn about [Full SFT](./sft-customization-job) for direct supervised fine-tuning", + "source_html": "

Interpreting ROUGE Scores:

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
MetricMeasures
ROUGE-1Unigram overlap between prediction and reference
ROUGE-2Bigram overlap (captures phrase-level similarity)
ROUGE-LLongest common subsequence (captures sentence structure)
ROUGE-LsumROUGE-L computed over full summaries
\n

What to expect:

\n
    \n
  • The base student (1B, no training) provides a lower bound since it has not seen the task data
  • \n
  • The distilled student (1B, KD) should significantly outperform the base student, demonstrating the knowledge transferred from the 3B teacher
  • \n
  • If the distilled student scores are not much higher than the baseline, try increasing distillation_temperature, adjusting distillation_ratio, or training for more epochs
  • \n
\n
\n

Hyperparameters

\n

For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the Hyperparameter Reference.

\n
\n

Troubleshooting

\n

Job fails during model download:

\n
    \n
  • Verify authentication secrets are configured (refer to Managing Secrets)
  • \n
  • For gated HuggingFace models (Llama, Gemma), accept the license on the model page
  • \n
  • Check both model (student) and teacher_model URNs are correct
  • \n
  • Ensure both model entities exist: client.models.retrieve(name=..., workspace="default")
  • \n
\n

Job fails with OOM (Out of Memory) error:

\n

KD loads both models, so OOM is more likely than with SFT:

\n
    \n
  1. First try: Use teacher_precision="bf16" to reduce teacher memory
  2. \n
  3. Still OOM: Reduce micro_batch_size to 1
  4. \n
  5. Still OOM: Reduce global_batch_size and max_seq_length
  6. \n
  7. Last resort: Increase num_gpus_per_node
  8. \n
\n

No chat template / /chat/completions fails:

\n
    \n
  • Use Instruct model variants (e.g., Llama-3.2-1B-Instruct) instead of base models (Llama-3.2-1B). Base models do not include a chat template in their tokenizer, so the output model will also lack one.
  • \n
\n

Distilled model quality is poor:

\n
    \n
  • Increase distillation_temperature (try 2.0–5.0) to transfer more nuanced knowledge
  • \n
  • Adjust distillation_ratio—if dataset labels are high-quality, lower the ratio; if the teacher is strong, raise it
  • \n
  • Increase epochs or max_steps for more training
  • \n
  • Verify teacher and student share the same vocabulary
  • \n
\n

Vocabulary mismatch error:

\n
    \n
  • Teacher and student must use the same tokenizer. Use models from the same family (e.g., Llama 3.2 1B Instruct + Llama 3.2 3B Instruct)
  • \n
\n

Deployment fails:

\n
    \n
  • Verify output model exists: client.models.retrieve(name=DISTILLED_STUDENT_NAME, workspace="default")
  • \n
  • Check deployment logs: client.inference.deployments.get_logs(name=deployment.name, workspace="default")
  • \n
  • The distilled model has the same size as the student, so GPU requirements match the student model
  • \n
\n

Next Steps

\n\n" } ] } \ No newline at end of file diff --git a/docs/fern/components/notebooks/distillation-customization-job.ts b/docs/fern/components/notebooks/distillation-customization-job.ts index e6db736617..5190b821cd 100644 --- a/docs/fern/components/notebooks/distillation-customization-job.ts +++ b/docs/fern/components/notebooks/distillation-customization-job.ts @@ -1,7 +1,9 @@ -// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -// SPDX-License-Identifier: Apache-2.0 - -/** Auto-generated by ipynb-to-fern-json.py - do not edit */ +/** + * SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + * + * Auto-generated by ipynb-to-fern-json.py - do not edit manually. + */ export default { cells: [ { "type": "markdown", @@ -10,8 +12,8 @@ export default { cells: [ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)\n3. **Installed evaluation dependencies:**\n\n```sh\npip install evaluate rouge_score datasets\n```", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
  5. Installed evaluation dependencies:
  6. \n
\n
pip install evaluate rouge_score datasets\n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n3. **Installed evaluation dependencies:**\n\n```sh\npip install evaluate rouge_score datasets\n```", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (PyPI wrapper: pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  4. \n
  5. Installed evaluation dependencies:
  6. \n
\n
pip install evaluate rouge_score datasets\n
\n" }, { "type": "markdown", @@ -37,9 +39,9 @@ export default { cells: [ }, { "type": "code", - "source": "DATASET_NAME = \"kd-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"Knowledge distillation training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\nprint(\"Uploaded files:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", + "source": "DATASET_NAME = \"kd-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"Knowledge distillation training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=f\"{DATASET_PATH}/\",\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\nprint(\"Uploaded files:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", "language": "python", - "source_html": "DATASET_NAME = "kd-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="Knowledge distillation training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\nprint("Uploaded files:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" + "source_html": "DATASET_NAME = "kd-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="Knowledge distillation training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=f"{DATASET_PATH}/",\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\nprint("Uploaded files:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" }, { "type": "markdown", @@ -70,15 +72,15 @@ export default { cells: [ }, { "type": "code", - "source": "from nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n DistillationTrainingParam,\n ParallelismParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\n\nTEACHER_JOB_NAME = f\"teacher-sft-job-{job_suffix}\"\n\nteacher_job = client.customization.jobs.create(\n name=TEACHER_JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{teacher_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nTRAINED_TEACHER_NAME = teacher_job.spec.output.name\nprint(f\"Teacher training job: {teacher_job.name}\")\nprint(f\"Output teacher model: {TRAINED_TEACHER_NAME}\")", + "source": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\n\nTEACHER_JOB_NAME = f\"teacher-sft-job-{job_suffix}\"\nTEACHER_OUTPUT_NAME = f\"teacher-model-{job_suffix}\"\n\nteacher_spec = AutomodelJobInput(\n model=f\"default/{teacher_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"all_weights\",\n \"max_seq_length\": 2048,\n },\n schedule={\"epochs\": 1},\n batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": TEACHER_OUTPUT_NAME},\n)\n\nteacher_job = client.customization.automodel.jobs.create(\n spec=teacher_spec, workspace=\"default\", name=TEACHER_JOB_NAME\n)\n\nTRAINED_TEACHER_NAME = TEACHER_OUTPUT_NAME\nprint(f\"Teacher training job: {teacher_job.job.name}\")\nprint(f\"Output teacher model: {TRAINED_TEACHER_NAME}\")", "language": "python", - "source_html": "from nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n DistillationTrainingParam,\n ParallelismParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\n\nTEACHER_JOB_NAME = f"teacher-sft-job-{job_suffix}"\n\nteacher_job = client.customization.jobs.create(\n name=TEACHER_JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{teacher_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nTRAINED_TEACHER_NAME = teacher_job.spec.output.name\nprint(f"Teacher training job: {teacher_job.name}")\nprint(f"Output teacher model: {TRAINED_TEACHER_NAME}")\n" + "source_html": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\n\nTEACHER_JOB_NAME = f"teacher-sft-job-{job_suffix}"\nTEACHER_OUTPUT_NAME = f"teacher-model-{job_suffix}"\n\nteacher_spec = AutomodelJobInput(\n model=f"default/{teacher_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "all_weights",\n "max_seq_length": 2048,\n },\n schedule={"epochs": 1},\n batch={"global_batch_size": 64, "micro_batch_size": 1},\n optimizer={"learning_rate": 5e-5},\n parallelism={"num_gpus_per_node": 1},\n output={"name": TEACHER_OUTPUT_NAME},\n)\n\nteacher_job = client.customization.automodel.jobs.create(\n spec=teacher_spec, workspace="default", name=TEACHER_JOB_NAME\n)\n\nTRAINED_TEACHER_NAME = TEACHER_OUTPUT_NAME\nprint(f"Teacher training job: {teacher_job.job.name}")\nprint(f"Output teacher model: {TRAINED_TEACHER_NAME}")\n" }, { "type": "code", - "source": "from IPython.display import clear_output\n\n\ndef wait_for_job(job_name: str):\n \"\"\"Poll job status until completion.\"\"\"\n while True:\n status = client.customization.jobs.get_status(name=job_name, workspace=\"default\")\n clear_output(wait=True)\n print(f\"Job: {job_name}\")\n print(f\"Status: {status.status}\")\n\n for job_step in status.steps or []:\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n details = task.status_details or {}\n step = details.get(\"step\")\n max_steps = details.get(\"max_steps\")\n if step is not None and max_steps is not None:\n print(f\"Progress: Step {step}/{max_steps} ({step / max_steps * 100:.1f}%)\")\n phase = details.get(\"phase\")\n if phase:\n print(f\"Phase: {phase}\")\n break\n break\n\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished: {status.status}\")\n return status\n\n time.sleep(10)\n\n\nteacher_status = wait_for_job(TEACHER_JOB_NAME)", + "source": "from IPython.display import clear_output\n\n\ndef wait_for_job(job_name: str):\n \"\"\"Poll job status until completion.\"\"\"\n while True:\n status = client.jobs.get_status(name=job_name, workspace=\"default\")\n clear_output(wait=True)\n print(f\"Job: {job_name}\")\n print(f\"Status: {status.status}\")\n\n for job_step in status.steps or []:\n if job_step.name == \"training\":\n for task in job_step.tasks or []:\n details = task.status_details or {}\n step = details.get(\"step\")\n max_steps = details.get(\"max_steps\")\n if step is not None and max_steps is not None:\n print(f\"Progress: Step {step}/{max_steps} ({step / max_steps * 100:.1f}%)\")\n phase = details.get(\"phase\")\n if phase:\n print(f\"Phase: {phase}\")\n break\n break\n\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished: {status.status}\")\n return status\n\n time.sleep(10)\n\n\nteacher_status = wait_for_job(TEACHER_JOB_NAME)\nassert teacher_status.status == \"completed\"", "language": "python", - "source_html": "from IPython.display import clear_output\n\n\ndef wait_for_job(job_name: str):\n """Poll job status until completion."""\n while True:\n status = client.customization.jobs.get_status(name=job_name, workspace="default")\n clear_output(wait=True)\n print(f"Job: {job_name}")\n print(f"Status: {status.status}")\n\n for job_step in status.steps or []:\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n details = task.status_details or {}\n step = details.get("step")\n max_steps = details.get("max_steps")\n if step is not None and max_steps is not None:\n print(f"Progress: Step {step}/{max_steps} ({step / max_steps * 100:.1f}%)")\n phase = details.get("phase")\n if phase:\n print(f"Phase: {phase}")\n break\n break\n\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished: {status.status}")\n return status\n\n time.sleep(10)\n\n\nteacher_status = wait_for_job(TEACHER_JOB_NAME)\n" + "source_html": "from IPython.display import clear_output\n\n\ndef wait_for_job(job_name: str):\n """Poll job status until completion."""\n while True:\n status = client.jobs.get_status(name=job_name, workspace="default")\n clear_output(wait=True)\n print(f"Job: {job_name}")\n print(f"Status: {status.status}")\n\n for job_step in status.steps or []:\n if job_step.name == "training":\n for task in job_step.tasks or []:\n details = task.status_details or {}\n step = details.get("step")\n max_steps = details.get("max_steps")\n if step is not None and max_steps is not None:\n print(f"Progress: Step {step}/{max_steps} ({step / max_steps * 100:.1f}%)")\n phase = details.get("phase")\n if phase:\n print(f"Phase: {phase}")\n break\n break\n\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished: {status.status}")\n return status\n\n time.sleep(10)\n\n\nteacher_status = wait_for_job(TEACHER_JOB_NAME)\nassert teacher_status.status == "completed"\n" }, { "type": "markdown", @@ -87,15 +89,15 @@ export default { cells: [ }, { "type": "code", - "source": "from nemo_platform.types.inference import NIMDeploymentParam\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f\"baseline-student-cfg-{baseline_suffix}\"\nBASELINE_DEPLOYMENT_NAME = f\"baseline-student-{baseline_suffix}\"\n\nbaseline_deployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n image_tag=\"1.15.5\",\n gpu=1,\n model_name=student_model.name,\n model_namespace=\"default\",\n additional_envs={\"NIM_MODEL_PROFILE\": \"vllm\"}\n ),\n)\n\nbaseline_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_deployment_config.name\n)\n\nprint(f\"Baseline student deployment: {baseline_deployment.name}\")", + "source": "baseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f\"baseline-student-cfg-{baseline_suffix}\"\nBASELINE_DEPLOYMENT_NAME = f\"baseline-student-{baseline_suffix}\"\n\nbaseline_deployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_CONFIG,\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": student_model.name,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n },\n)\n\nbaseline_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_deployment_config.name\n)\n\nprint(f\"Baseline student deployment: {baseline_deployment.name}\")", "language": "python", - "source_html": "from nemo_platform.types.inference import NIMDeploymentParam\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f"baseline-student-cfg-{baseline_suffix}"\nBASELINE_DEPLOYMENT_NAME = f"baseline-student-{baseline_suffix}"\n\nbaseline_deployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name="nvcr.io/nim/nvidia/llm-nim",\n image_tag="1.15.5",\n gpu=1,\n model_name=student_model.name,\n model_namespace="default",\n additional_envs={"NIM_MODEL_PROFILE": "vllm"}\n ),\n)\n\nbaseline_deployment = client.inference.deployments.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_deployment_config.name\n)\n\nprint(f"Baseline student deployment: {baseline_deployment.name}")\n" + "source_html": "baseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f"baseline-student-cfg-{baseline_suffix}"\nBASELINE_DEPLOYMENT_NAME = f"baseline-student-{baseline_suffix}"\n\nbaseline_deployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_CONFIG,\n engine="vllm",\n model_spec={\n "model_namespace": "default",\n "model_name": student_model.name,\n },\n executor_config={\n "gpu": 1,\n "image_name": "vllm/vllm-openai",\n "image_tag": "v0.22.1",\n },\n)\n\nbaseline_deployment = client.inference.deployments.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_deployment_config.name\n)\n\nprint(f"Baseline student deployment: {baseline_deployment.name}")\n" }, { "type": "code", - "source": "def wait_for_deployment(deployment_name: str, timeout_minutes: int = 30):\n \"\"\"Poll deployment until ready.\"\"\"\n start = time.time()\n timeout = timeout_minutes * 60\n while True:\n dep = client.inference.deployments.retrieve(name=deployment_name, workspace=\"default\")\n elapsed = time.time() - start\n clear_output(wait=True)\n print(f\"Deployment: {deployment_name}\")\n print(f\"Status: {dep.status}\")\n print(f\"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s\")\n\n if dep.status == \"READY\":\n print(\"\\nDeployment is ready!\")\n return dep\n if dep.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n print(f\"\\nDeployment failed: {dep.status}\")\n return dep\n if elapsed > timeout:\n print(f\"\\nTimeout ({timeout_minutes}m). Check status manually.\")\n return dep\n time.sleep(15)\n\n\nwait_for_deployment(BASELINE_DEPLOYMENT_NAME)", + "source": "def wait_for_deployment(deployment_name: str, timeout_minutes: int = 30):\n \"\"\"Poll deployment until ready.\"\"\"\n start = time.time()\n timeout = timeout_minutes * 60\n while True:\n dep = client.inference.deployments.retrieve(name=deployment_name, workspace=\"default\")\n elapsed = time.time() - start\n clear_output(wait=True)\n print(f\"Deployment: {deployment_name}\")\n print(f\"Status: {dep.status}\")\n print(f\"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s\")\n\n if dep.status == \"READY\":\n print(\"\\nDeployment is ready!\")\n return dep\n if dep.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n print(f\"\\nDeployment failed: {dep.status}\")\n return dep\n if elapsed > timeout:\n print(f\"\\nTimeout ({timeout_minutes}m). Check status manually.\")\n return dep\n time.sleep(15)\n\n\ndep_status = wait_for_deployment(BASELINE_DEPLOYMENT_NAME)\nassert dep_status.status == \"READY\"", "language": "python", - "source_html": "def wait_for_deployment(deployment_name: str, timeout_minutes: int = 30):\n """Poll deployment until ready."""\n start = time.time()\n timeout = timeout_minutes * 60\n while True:\n dep = client.inference.deployments.retrieve(name=deployment_name, workspace="default")\n elapsed = time.time() - start\n clear_output(wait=True)\n print(f"Deployment: {deployment_name}")\n print(f"Status: {dep.status}")\n print(f"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s")\n\n if dep.status == "READY":\n print("\\nDeployment is ready!")\n return dep\n if dep.status in ("FAILED", "ERROR", "TERMINATED", "LOST"):\n print(f"\\nDeployment failed: {dep.status}")\n return dep\n if elapsed > timeout:\n print(f"\\nTimeout ({timeout_minutes}m). Check status manually.")\n return dep\n time.sleep(15)\n\n\nwait_for_deployment(BASELINE_DEPLOYMENT_NAME)\n" + "source_html": "def wait_for_deployment(deployment_name: str, timeout_minutes: int = 30):\n """Poll deployment until ready."""\n start = time.time()\n timeout = timeout_minutes * 60\n while True:\n dep = client.inference.deployments.retrieve(name=deployment_name, workspace="default")\n elapsed = time.time() - start\n clear_output(wait=True)\n print(f"Deployment: {deployment_name}")\n print(f"Status: {dep.status}")\n print(f"Elapsed: {int(elapsed // 60)}m {int(elapsed % 60)}s")\n\n if dep.status == "READY":\n print("\\nDeployment is ready!")\n return dep\n if dep.status in ("FAILED", "ERROR", "TERMINATED", "LOST"):\n print(f"\\nDeployment failed: {dep.status}")\n return dep\n if elapsed > timeout:\n print(f"\\nTimeout ({timeout_minutes}m). Check status manually.")\n return dep\n time.sleep(15)\n\n\ndep_status = wait_for_deployment(BASELINE_DEPLOYMENT_NAME)\nassert dep_status.status == "READY"\n" }, { "type": "markdown", @@ -121,9 +123,9 @@ export default { cells: [ }, { "type": "code", - "source": "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace=\"default\")\nprint(f\"Deleted baseline deployment: {BASELINE_DEPLOYMENT_NAME}\")\n\n# wait for deployment to be deleted\ntime.sleep(60)\n\nclient.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace=\"default\")\nprint(f\"Deleted baseline deployment config: {BASELINE_DEPLOYMENT_CONFIG}\")", + "source": "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace=\"default\")\nprint(f\"Deleted baseline deployment: {BASELINE_DEPLOYMENT_NAME}\")\n\nif not client.models.wait_for_status(\n deployment_name=BASELINE_DEPLOYMENT_NAME,\n desired_status=\"DELETED\",\n workspace=\"default\",\n timeout=600,\n):\n raise TimeoutError(\n f\"Deployment {BASELINE_DEPLOYMENT_NAME} was not deleted within timeout\"\n )\n\nclient.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace=\"default\")\nprint(f\"Deleted baseline deployment config: {BASELINE_DEPLOYMENT_CONFIG}\")", "language": "python", - "source_html": "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace="default")\nprint(f"Deleted baseline deployment: {BASELINE_DEPLOYMENT_NAME}")\n\n# wait for deployment to be deleted\ntime.sleep(60)\n\nclient.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace="default")\nprint(f"Deleted baseline deployment config: {BASELINE_DEPLOYMENT_CONFIG}")\n" + "source_html": "client.inference.deployments.delete(name=BASELINE_DEPLOYMENT_NAME, workspace="default")\nprint(f"Deleted baseline deployment: {BASELINE_DEPLOYMENT_NAME}")\n\nif not client.models.wait_for_status(\n deployment_name=BASELINE_DEPLOYMENT_NAME,\n desired_status="DELETED",\n workspace="default",\n timeout=600,\n):\n raise TimeoutError(\n f"Deployment {BASELINE_DEPLOYMENT_NAME} was not deleted within timeout"\n )\n\nclient.inference.deployment_configs.delete(name=BASELINE_DEPLOYMENT_CONFIG, workspace="default")\nprint(f"Deleted baseline deployment config: {BASELINE_DEPLOYMENT_CONFIG}")\n" }, { "type": "markdown", @@ -137,9 +139,9 @@ export default { cells: [ }, { "type": "code", - "source": "KD_JOB_NAME = f\"my-kd-job-{job_suffix}\"\n\nkd_job = client.customization.jobs.create(\n name=KD_JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{student_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=DistillationTrainingParam(\n type=\"distillation\",\n teacher_model=f\"default/{TRAINED_TEACHER_NAME}\",\n teacher_precision=\"bf16\",\n distillation_ratio=0.5,\n distillation_temperature=2.0,\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nDISTILLED_STUDENT_NAME = kd_job.spec.output.name\nprint(f\"Distillation job: {kd_job.name}\")\nprint(f\"Output student model: {DISTILLED_STUDENT_NAME}\")", + "source": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\nKD_JOB_NAME = f\"my-kd-job-{job_suffix}\"\nKD_OUTPUT_NAME = f\"kd-student-{job_suffix}\"\n\nkd_spec = AutomodelJobInput(\n model=f\"default/{student_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"distillation\",\n \"finetuning_type\": \"all_weights\",\n \"teacher_model\": f\"default/{TRAINED_TEACHER_NAME}\",\n \"teacher_precision\": \"bf16\",\n \"distillation_ratio\": 0.5,\n \"distillation_temperature\": 2.0,\n \"max_seq_length\": 2048,\n },\n schedule={\"epochs\": 1},\n batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": KD_OUTPUT_NAME},\n)\n\nkd_job = client.customization.automodel.jobs.create(\n spec=kd_spec, workspace=\"default\", name=KD_JOB_NAME\n)\n\nDISTILLED_STUDENT_NAME = KD_OUTPUT_NAME\nprint(f\"Distillation job: {kd_job.job.name}\")\nprint(f\"Output student model: {DISTILLED_STUDENT_NAME}\")", "language": "python", - "source_html": "KD_JOB_NAME = f"my-kd-job-{job_suffix}"\n\nkd_job = client.customization.jobs.create(\n name=KD_JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{student_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=DistillationTrainingParam(\n type="distillation",\n teacher_model=f"default/{TRAINED_TEACHER_NAME}",\n teacher_precision="bf16",\n distillation_ratio=0.5,\n distillation_temperature=2.0,\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nDISTILLED_STUDENT_NAME = kd_job.spec.output.name\nprint(f"Distillation job: {kd_job.name}")\nprint(f"Output student model: {DISTILLED_STUDENT_NAME}")\n" + "source_html": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\nKD_JOB_NAME = f"my-kd-job-{job_suffix}"\nKD_OUTPUT_NAME = f"kd-student-{job_suffix}"\n\nkd_spec = AutomodelJobInput(\n model=f"default/{student_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "distillation",\n "finetuning_type": "all_weights",\n "teacher_model": f"default/{TRAINED_TEACHER_NAME}",\n "teacher_precision": "bf16",\n "distillation_ratio": 0.5,\n "distillation_temperature": 2.0,\n "max_seq_length": 2048,\n },\n schedule={"epochs": 1},\n batch={"global_batch_size": 64, "micro_batch_size": 1},\n optimizer={"learning_rate": 5e-5},\n parallelism={"num_gpus_per_node": 1},\n output={"name": KD_OUTPUT_NAME},\n)\n\nkd_job = client.customization.automodel.jobs.create(\n spec=kd_spec, workspace="default", name=KD_JOB_NAME\n)\n\nDISTILLED_STUDENT_NAME = KD_OUTPUT_NAME\nprint(f"Distillation job: {kd_job.job.name}")\nprint(f"Output student model: {DISTILLED_STUDENT_NAME}")\n" }, { "type": "markdown", @@ -148,9 +150,9 @@ export default { cells: [ }, { "type": "code", - "source": "kd_status = wait_for_job(KD_JOB_NAME)", + "source": "kd_status = wait_for_job(KD_JOB_NAME)\nassert kd_status.status == \"completed\"", "language": "python", - "source_html": "kd_status = wait_for_job(KD_JOB_NAME)\n" + "source_html": "kd_status = wait_for_job(KD_JOB_NAME)\nassert kd_status.status == "completed"\n" }, { "type": "markdown", @@ -159,9 +161,9 @@ export default { cells: [ }, { "type": "code", - "source": "deploy_suffix_2 = uuid.uuid4().hex[:4]\nSTUDENT_DEPLOYMENT_CONFIG = f\"kd-student-deploy-cfg-{deploy_suffix_2}\"\nSTUDENT_DEPLOYMENT_NAME = f\"kd-student-deploy-{deploy_suffix_2}\"\n\nstudent_deployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=STUDENT_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n image_tag=\"1.15.5\",\n gpu=1,\n model_name=DISTILLED_STUDENT_NAME,\n model_namespace=\"default\",\n additional_envs={\"NIM_MODEL_PROFILE\": \"vllm\"},\n ),\n)\n\nstudent_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=STUDENT_DEPLOYMENT_NAME,\n config=student_deployment_config.name\n)\n\nprint(f\"Student deployment: {student_deployment.name}\")", + "source": "deploy_suffix_2 = uuid.uuid4().hex[:4]\nSTUDENT_DEPLOYMENT_CONFIG = f\"kd-student-deploy-cfg-{deploy_suffix_2}\"\nSTUDENT_DEPLOYMENT_NAME = f\"kd-student-deploy-{deploy_suffix_2}\"\n\nstudent_deployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=STUDENT_DEPLOYMENT_CONFIG,\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": DISTILLED_STUDENT_NAME,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n },\n)\n\nstudent_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=STUDENT_DEPLOYMENT_NAME,\n config=student_deployment_config.name\n)\n\nprint(f\"Student deployment: {student_deployment.name}\")", "language": "python", - "source_html": "deploy_suffix_2 = uuid.uuid4().hex[:4]\nSTUDENT_DEPLOYMENT_CONFIG = f"kd-student-deploy-cfg-{deploy_suffix_2}"\nSTUDENT_DEPLOYMENT_NAME = f"kd-student-deploy-{deploy_suffix_2}"\n\nstudent_deployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=STUDENT_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name="nvcr.io/nim/nvidia/llm-nim",\n image_tag="1.15.5",\n gpu=1,\n model_name=DISTILLED_STUDENT_NAME,\n model_namespace="default",\n additional_envs={"NIM_MODEL_PROFILE": "vllm"},\n ),\n)\n\nstudent_deployment = client.inference.deployments.create(\n workspace="default",\n name=STUDENT_DEPLOYMENT_NAME,\n config=student_deployment_config.name\n)\n\nprint(f"Student deployment: {student_deployment.name}")\n" + "source_html": "deploy_suffix_2 = uuid.uuid4().hex[:4]\nSTUDENT_DEPLOYMENT_CONFIG = f"kd-student-deploy-cfg-{deploy_suffix_2}"\nSTUDENT_DEPLOYMENT_NAME = f"kd-student-deploy-{deploy_suffix_2}"\n\nstudent_deployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=STUDENT_DEPLOYMENT_CONFIG,\n engine="vllm",\n model_spec={\n "model_namespace": "default",\n "model_name": DISTILLED_STUDENT_NAME,\n },\n executor_config={\n "gpu": 1,\n "image_name": "vllm/vllm-openai",\n "image_tag": "v0.22.1",\n },\n)\n\nstudent_deployment = client.inference.deployments.create(\n workspace="default",\n name=STUDENT_DEPLOYMENT_NAME,\n config=student_deployment_config.name\n)\n\nprint(f"Student deployment: {student_deployment.name}")\n" }, { "type": "code", @@ -199,7 +201,7 @@ export default { cells: [ }, { "type": "markdown", - "source": "**Interpreting ROUGE Scores:**\n\n| Metric | Measures |\n|--------|----------|\n| **ROUGE-1** | Unigram overlap between prediction and reference |\n| **ROUGE-2** | Bigram overlap (captures phrase-level similarity) |\n| **ROUGE-L** | Longest common subsequence (captures sentence structure) |\n| **ROUGE-Lsum** | ROUGE-L computed over full summaries |\n\n**What to expect:**\n- The base student (1B, no training) provides a lower bound since it has not seen the task data\n- The distilled student (1B, KD) should significantly outperform the base student, demonstrating the knowledge transferred from the 3B teacher\n- If the distilled student scores are not much higher than the baseline, try increasing `distillation_temperature`, adjusting `distillation_ratio`, or training for more epochs\n\n---\n\n## Hyperparameters\n\nFor detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n\n---\n\n## Troubleshooting\n\n**Job fails during model download:**\n- Verify authentication secrets are configured (refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n- Check both `model` (student) and `teacher_model` URNs are correct\n- Ensure both model entities exist: `client.models.retrieve(name=..., workspace=\"default\")`\n\n**Job fails with OOM (Out of Memory) error:**\n\nKD loads both models, so OOM is more likely than with SFT:\n1. **First try:** Use `teacher_precision=\"bf16\"` to reduce teacher memory\n2. **Still OOM:** Reduce `micro_batch_size` to 1\n3. **Still OOM:** Reduce `batch_size` and `max_seq_length`\n4. **Last resort:** Increase `num_gpus_per_node`\n\n**No chat template / `/chat/completions` fails:**\n- Use Instruct model variants (e.g., `Llama-3.2-1B-Instruct`) instead of base models (`Llama-3.2-1B`). Base models do not include a chat template in their tokenizer, so the output model will also lack one.\n\n**Distilled model quality is poor:**\n- Increase `distillation_temperature` (try 2.0–5.0) to transfer more nuanced knowledge\n- Adjust `distillation_ratio`—if dataset labels are high-quality, lower the ratio; if the teacher is strong, raise it\n- Increase `epochs` or `max_steps` for more training\n- Verify teacher and student share the same vocabulary\n\n**Vocabulary mismatch error:**\n- Teacher and student must use the same tokenizer. Use models from the same family (e.g., Llama 3.2 1B Instruct + Llama 3.2 3B Instruct)\n\n**Deployment fails:**\n- Verify output model exists: `client.models.retrieve(name=job.spec.output.name, workspace=\"default\")`\n- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n- The distilled model has the same size as the student, so GPU requirements match the student model\n\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning\n- Learn about [Full SFT](./sft-customization-job) for direct supervised fine-tuning", - "source_html": "

Interpreting ROUGE Scores:

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
MetricMeasures
ROUGE-1Unigram overlap between prediction and reference
ROUGE-2Bigram overlap (captures phrase-level similarity)
ROUGE-LLongest common subsequence (captures sentence structure)
ROUGE-LsumROUGE-L computed over full summaries
\n

What to expect:

\n
    \n
  • The base student (1B, no training) provides a lower bound since it has not seen the task data
  • \n
  • The distilled student (1B, KD) should significantly outperform the base student, demonstrating the knowledge transferred from the 3B teacher
  • \n
  • If the distilled student scores are not much higher than the baseline, try increasing distillation_temperature, adjusting distillation_ratio, or training for more epochs
  • \n
\n
\n

Hyperparameters

\n

For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the Hyperparameter Reference.

\n
\n

Troubleshooting

\n

Job fails during model download:

\n
    \n
  • Verify authentication secrets are configured (refer to Managing Secrets)
  • \n
  • For gated HuggingFace models (Llama, Gemma), accept the license on the model page
  • \n
  • Check both model (student) and teacher_model URNs are correct
  • \n
  • Ensure both model entities exist: client.models.retrieve(name=..., workspace="default")
  • \n
\n

Job fails with OOM (Out of Memory) error:

\n

KD loads both models, so OOM is more likely than with SFT:

\n
    \n
  1. First try: Use teacher_precision="bf16" to reduce teacher memory
  2. \n
  3. Still OOM: Reduce micro_batch_size to 1
  4. \n
  5. Still OOM: Reduce batch_size and max_seq_length
  6. \n
  7. Last resort: Increase num_gpus_per_node
  8. \n
\n

No chat template / /chat/completions fails:

\n
    \n
  • Use Instruct model variants (e.g., Llama-3.2-1B-Instruct) instead of base models (Llama-3.2-1B). Base models do not include a chat template in their tokenizer, so the output model will also lack one.
  • \n
\n

Distilled model quality is poor:

\n
    \n
  • Increase distillation_temperature (try 2.0–5.0) to transfer more nuanced knowledge
  • \n
  • Adjust distillation_ratio—if dataset labels are high-quality, lower the ratio; if the teacher is strong, raise it
  • \n
  • Increase epochs or max_steps for more training
  • \n
  • Verify teacher and student share the same vocabulary
  • \n
\n

Vocabulary mismatch error:

\n
    \n
  • Teacher and student must use the same tokenizer. Use models from the same family (e.g., Llama 3.2 1B Instruct + Llama 3.2 3B Instruct)
  • \n
\n

Deployment fails:

\n
    \n
  • Verify output model exists: client.models.retrieve(name=job.spec.output.name, workspace="default")
  • \n
  • Check deployment logs: client.inference.deployments.get_logs(name=deployment.name, workspace="default")
  • \n
  • The distilled model has the same size as the student, so GPU requirements match the student model
  • \n
\n

Next Steps

\n\n" + "source": "**Interpreting ROUGE Scores:**\n\n| Metric | Measures |\n|--------|----------|\n| **ROUGE-1** | Unigram overlap between prediction and reference |\n| **ROUGE-2** | Bigram overlap (captures phrase-level similarity) |\n| **ROUGE-L** | Longest common subsequence (captures sentence structure) |\n| **ROUGE-Lsum** | ROUGE-L computed over full summaries |\n\n**What to expect:**\n- The base student (1B, no training) provides a lower bound since it has not seen the task data\n- The distilled student (1B, KD) should significantly outperform the base student, demonstrating the knowledge transferred from the 3B teacher\n- If the distilled student scores are not much higher than the baseline, try increasing `distillation_temperature`, adjusting `distillation_ratio`, or training for more epochs\n\n---\n\n## Hyperparameters\n\nFor detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n\n---\n\n## Troubleshooting\n\n**Job fails during model download:**\n- Verify authentication secrets are configured (refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n- Check both `model` (student) and `teacher_model` URNs are correct\n- Ensure both model entities exist: `client.models.retrieve(name=..., workspace=\"default\")`\n\n**Job fails with OOM (Out of Memory) error:**\n\nKD loads both models, so OOM is more likely than with SFT:\n1. **First try:** Use `teacher_precision=\"bf16\"` to reduce teacher memory\n2. **Still OOM:** Reduce `micro_batch_size` to 1\n3. **Still OOM:** Reduce `global_batch_size` and `max_seq_length`\n4. **Last resort:** Increase `num_gpus_per_node`\n\n**No chat template / `/chat/completions` fails:**\n- Use Instruct model variants (e.g., `Llama-3.2-1B-Instruct`) instead of base models (`Llama-3.2-1B`). Base models do not include a chat template in their tokenizer, so the output model will also lack one.\n\n**Distilled model quality is poor:**\n- Increase `distillation_temperature` (try 2.0–5.0) to transfer more nuanced knowledge\n- Adjust `distillation_ratio`—if dataset labels are high-quality, lower the ratio; if the teacher is strong, raise it\n- Increase `epochs` or `max_steps` for more training\n- Verify teacher and student share the same vocabulary\n\n**Vocabulary mismatch error:**\n- Teacher and student must use the same tokenizer. Use models from the same family (e.g., Llama 3.2 1B Instruct + Llama 3.2 3B Instruct)\n\n**Deployment fails:**\n- Verify output model exists: `client.models.retrieve(name=DISTILLED_STUDENT_NAME, workspace=\"default\")`\n- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n- The distilled model has the same size as the student, so GPU requirements match the student model\n\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning\n- Learn about [Full SFT](./sft-customization-job) for direct supervised fine-tuning", + "source_html": "

Interpreting ROUGE Scores:

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
MetricMeasures
ROUGE-1Unigram overlap between prediction and reference
ROUGE-2Bigram overlap (captures phrase-level similarity)
ROUGE-LLongest common subsequence (captures sentence structure)
ROUGE-LsumROUGE-L computed over full summaries
\n

What to expect:

\n
    \n
  • The base student (1B, no training) provides a lower bound since it has not seen the task data
  • \n
  • The distilled student (1B, KD) should significantly outperform the base student, demonstrating the knowledge transferred from the 3B teacher
  • \n
  • If the distilled student scores are not much higher than the baseline, try increasing distillation_temperature, adjusting distillation_ratio, or training for more epochs
  • \n
\n
\n

Hyperparameters

\n

For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the Hyperparameter Reference.

\n
\n

Troubleshooting

\n

Job fails during model download:

\n
    \n
  • Verify authentication secrets are configured (refer to Managing Secrets)
  • \n
  • For gated HuggingFace models (Llama, Gemma), accept the license on the model page
  • \n
  • Check both model (student) and teacher_model URNs are correct
  • \n
  • Ensure both model entities exist: client.models.retrieve(name=..., workspace="default")
  • \n
\n

Job fails with OOM (Out of Memory) error:

\n

KD loads both models, so OOM is more likely than with SFT:

\n
    \n
  1. First try: Use teacher_precision="bf16" to reduce teacher memory
  2. \n
  3. Still OOM: Reduce micro_batch_size to 1
  4. \n
  5. Still OOM: Reduce global_batch_size and max_seq_length
  6. \n
  7. Last resort: Increase num_gpus_per_node
  8. \n
\n

No chat template / /chat/completions fails:

\n
    \n
  • Use Instruct model variants (e.g., Llama-3.2-1B-Instruct) instead of base models (Llama-3.2-1B). Base models do not include a chat template in their tokenizer, so the output model will also lack one.
  • \n
\n

Distilled model quality is poor:

\n
    \n
  • Increase distillation_temperature (try 2.0–5.0) to transfer more nuanced knowledge
  • \n
  • Adjust distillation_ratio—if dataset labels are high-quality, lower the ratio; if the teacher is strong, raise it
  • \n
  • Increase epochs or max_steps for more training
  • \n
  • Verify teacher and student share the same vocabulary
  • \n
\n

Vocabulary mismatch error:

\n
    \n
  • Teacher and student must use the same tokenizer. Use models from the same family (e.g., Llama 3.2 1B Instruct + Llama 3.2 3B Instruct)
  • \n
\n

Deployment fails:

\n
    \n
  • Verify output model exists: client.models.retrieve(name=DISTILLED_STUDENT_NAME, workspace="default")
  • \n
  • Check deployment logs: client.inference.deployments.get_logs(name=deployment.name, workspace="default")
  • \n
  • The distilled model has the same size as the student, so GPU requirements match the student model
  • \n
\n

Next Steps

\n\n" } ] }; diff --git a/docs/fern/components/notebooks/dpo-customization-job.json b/docs/fern/components/notebooks/dpo-customization-job.json deleted file mode 100644 index c1ffcb3c49..0000000000 --- a/docs/fern/components/notebooks/dpo-customization-job.json +++ /dev/null @@ -1,203 +0,0 @@ -{ - "cells": [ - { - "type": "markdown", - "source": "\n\n\n# DPO Customization\n\nLearn how to use the NeMo Platform to create a DPO (Direct Preference Optimization) job using a custom dataset.\n\n## About\n\nDPO is an advanced fine-tuning technique for preference-based alignment. If you're new to fine-tuning, consider starting with [LoRA](./lora-customization-job) or [Full SFT](./sft-customization-job) tutorials first.\n\nDirect Preference Optimization (DPO) is an RL-free alignment algorithm that operates on preference data. Given a prompt and a pair of chosen and rejected responses, DPO aims to increase the probability of the chosen response and decrease the probability of the rejected response relative to a frozen reference model. The actor is initialized using the reference model. For more details, refer to the [DPO paper](https://arxiv.org/pdf/2305.18290).\n\nDPO shares similarities with Full SFT training workflows but differs in a few key ways:\n\n| Aspect | SFT (Supervised Fine-Tuning) | DPO (Direct Preference Optimization) |\n| --- | --- | --- |\n| Data Requirements | Labeled instruction-response pairs where the desired output is explicitly provided | Pairwise preference data, where for a given input, one response is explicitly preferred over another |\n| Learning Objective | Directly teaches the model to generate a specific \"correct\" response | Directly optimizes the model to align with human preferences by maximizing the probability of preferred responses and minimizing rejected ones, without needing an explicit reward model |\n| Alignment Focus | Aligns the model with the specific examples present in its training data | Aligns the model with broader human preferences, which can be more effective for subjective tasks or those without a single \"correct\" answer |\n| Computational Efficiency | Standard fine-tuning efficiency | More computationally efficient than SFT (especially when compared to full RLHF methods) as it bypasses the need to train a separate reward model |\n\n**What you can achieve with DPO:**\n- **Align with human preferences**: Directly optimize your model to produce outputs that align with subjective human preferences without requiring explicit reward modeling\n- **Refine response quality**: Improve helpfulness, harmlessness, honesty, and other nuanced qualities that are easier to compare than to define\n- **Control tone and style**: Adjust the model's communication style, verbosity, formality, and other subjective characteristics\n- **Implement safety guardrails**: Teach the model to avoid harmful or undesirable responses by training on preferred vs. rejected response pairs\n- **Optimize subjective tasks**: Excel at tasks where there are multiple acceptable answers but clear preferences exist (creative writing, dialogue, explanations)\n\n**When to choose DPO:**\n- **Subjective quality matters**: Your task involves style, tone, or other qualities where there's no single \"correct\" answer but clear preferences exist\n- **You have preference data**: You can collect pairwise comparisons (preferred vs. rejected responses) more easily than perfect labeled examples\n- **Refining existing capabilities**: You want to make targeted improvements to an already-trained model without major capability changes\n- **Complex evaluation**: Humans find it easier to compare which of two responses is better than to create the ideal response themselves (especially for multi-turn conversations, creative tasks, or nuanced outputs)\n- **Robust behavior changes**: You need more reliable behavior modification than prompting can provide, without the complexity of full RLHF\n- **Lower compute than RLHF**: You want human preference alignment but with simpler training that doesn't require reinforcement learning infrastructure\n\n**When to choose SFT:**\n- **Clear correct answers**: Your task has objectively correct outputs (code generation, structured data extraction, following specific formats)\n- **High-quality examples**: You have well-labeled input-output pairs that demonstrate exactly what the model should produce\n- **Imitation learning**: You want the model to closely mimic a specific style, format, or knowledge base from expert demonstrations\n- **Foundational capabilities**: You're establishing new task-specific capabilities before fine-tuning preferences (SFT is often done before DPO)\n- **Stable, predictable outputs**: You need consistent formatting or structure that's well-defined in your training examples\n- **Traditional NLP tasks**: Instruction following, translation, summarization, or classification where gold-standard labels exist", - "source_html": "\n\n

DPO Customization

\n

Learn how to use the NeMo Platform to create a DPO (Direct Preference Optimization) job using a custom dataset.

\n

About

\n

DPO is an advanced fine-tuning technique for preference-based alignment. If you're new to fine-tuning, consider starting with LoRA or Full SFT tutorials first.

\n

Direct Preference Optimization (DPO) is an RL-free alignment algorithm that operates on preference data. Given a prompt and a pair of chosen and rejected responses, DPO aims to increase the probability of the chosen response and decrease the probability of the rejected response relative to a frozen reference model. The actor is initialized using the reference model. For more details, refer to the DPO paper.

\n

DPO shares similarities with Full SFT training workflows but differs in a few key ways:

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
AspectSFT (Supervised Fine-Tuning)DPO (Direct Preference Optimization)
Data RequirementsLabeled instruction-response pairs where the desired output is explicitly providedPairwise preference data, where for a given input, one response is explicitly preferred over another
Learning ObjectiveDirectly teaches the model to generate a specific "correct" responseDirectly optimizes the model to align with human preferences by maximizing the probability of preferred responses and minimizing rejected ones, without needing an explicit reward model
Alignment FocusAligns the model with the specific examples present in its training dataAligns the model with broader human preferences, which can be more effective for subjective tasks or those without a single "correct" answer
Computational EfficiencyStandard fine-tuning efficiencyMore computationally efficient than SFT (especially when compared to full RLHF methods) as it bypasses the need to train a separate reward model
\n

What you can achieve with DPO:

\n
    \n
  • Align with human preferences: Directly optimize your model to produce outputs that align with subjective human preferences without requiring explicit reward modeling
  • \n
  • Refine response quality: Improve helpfulness, harmlessness, honesty, and other nuanced qualities that are easier to compare than to define
  • \n
  • Control tone and style: Adjust the model's communication style, verbosity, formality, and other subjective characteristics
  • \n
  • Implement safety guardrails: Teach the model to avoid harmful or undesirable responses by training on preferred vs. rejected response pairs
  • \n
  • Optimize subjective tasks: Excel at tasks where there are multiple acceptable answers but clear preferences exist (creative writing, dialogue, explanations)
  • \n
\n

When to choose DPO:

\n
    \n
  • Subjective quality matters: Your task involves style, tone, or other qualities where there's no single "correct" answer but clear preferences exist
  • \n
  • You have preference data: You can collect pairwise comparisons (preferred vs. rejected responses) more easily than perfect labeled examples
  • \n
  • Refining existing capabilities: You want to make targeted improvements to an already-trained model without major capability changes
  • \n
  • Complex evaluation: Humans find it easier to compare which of two responses is better than to create the ideal response themselves (especially for multi-turn conversations, creative tasks, or nuanced outputs)
  • \n
  • Robust behavior changes: You need more reliable behavior modification than prompting can provide, without the complexity of full RLHF
  • \n
  • Lower compute than RLHF: You want human preference alignment but with simpler training that doesn't require reinforcement learning infrastructure
  • \n
\n

When to choose SFT:

\n
    \n
  • Clear correct answers: Your task has objectively correct outputs (code generation, structured data extraction, following specific formats)
  • \n
  • High-quality examples: You have well-labeled input-output pairs that demonstrate exactly what the model should produce
  • \n
  • Imitation learning: You want the model to closely mimic a specific style, format, or knowledge base from expert demonstrations
  • \n
  • Foundational capabilities: You're establishing new task-specific capabilities before fine-tuning preferences (SFT is often done before DPO)
  • \n
  • Stable, predictable outputs: You need consistent formatting or structure that's well-defined in your training examples
  • \n
  • Traditional NLP tasks: Instruction following, translation, summarization, or classification where gold-standard labels exist
  • \n
\n" - }, - { - "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
\n" - }, - { - "type": "markdown", - "source": "## Quick Start\n\n### 1. Initialize SDK\n\nThe SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](../../get-started/quickstart.md) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable:\n\n```sh\nexport NMP_BASE_URL=\n```", - "source_html": "

Quick Start

\n

1. Initialize SDK

\n

The SDK needs to know your NeMo Platform server URL. By default, http://localhost:8080 is used in accordance with the Quickstart guide. If NeMo Platform is running at a custom location, you can override the URL by setting the NMP_BASE_URL environment variable:

\n
export NMP_BASE_URL=<YOUR_NMP_BASE_URL>\n
\n" - }, - { - "type": "code", - "source": "import json\nimport os\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n DpoTrainingParam,\n ParallelismParamsParam,\n)\n\nNMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace=\"default\"\n)", - "language": "python", - "source_html": "import json\nimport os\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n DpoTrainingParam,\n ParallelismParamsParam,\n)\n\nNMP_BASE_URL = os.environ.get("NMP_BASE_URL", "http://localhost:8080")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace="default"\n)\n" - }, - { - "type": "markdown", - "source": "### 2. Prepare Dataset\n\nCreate your data in JSONL format - one JSON object per line. The platform auto-detects your data format. Supported dataset formats are listed below.\n\n**Flexible Data Setup:**\n- **No validation file?** The platform automatically creates a 10% validation split\n- **Multiple files?** Upload to `training/` or `validation/` subdirectories—they'll be automatically merged\n- **Format detection:** Your data format is auto-detected at training time\n\nIn this tutorial the following dataset directory structure will be used:\n```\nmy_dataset\n`-- training.jsonl\n`-- validation.jsonl\n```", - "source_html": "

2. Prepare Dataset

\n

Create your data in JSONL format - one JSON object per line. The platform auto-detects your data format. Supported dataset formats are listed below.

\n

Flexible Data Setup:

\n
    \n
  • No validation file? The platform automatically creates a 10% validation split
  • \n
  • Multiple files? Upload to training/ or validation/ subdirectories—they'll be automatically merged
  • \n
  • Format detection: Your data format is auto-detected at training time
  • \n
\n

In this tutorial the following dataset directory structure will be used:

\n
my_dataset\n`-- training.jsonl\n`-- validation.jsonl\n
\n" - }, - { - "type": "markdown", - "source": "#### Binary Preference Format\nDPO training requires preference pairs with three fields:\n- **`prompt`**: The input prompt (can be a string or array of message objects)\n- **`chosen`**: The preferred response\n- **`rejected`**: The less preferred response", - "source_html": "

Binary Preference Format

\n

DPO training requires preference pairs with three fields:

\n
    \n
  • prompt: The input prompt (can be a string or array of message objects)
  • \n
  • chosen: The preferred response
  • \n
  • rejected: The less preferred response
  • \n
\n" - }, - { - "type": "code", - "source": "{\"prompt\": [{\"role\": \"user\", \"content\": \"What is the capital of France?\"}], \"chosen\": \"The capital of France is Paris. It is the largest city in France and serves as the country's political, economic, and cultural center.\", \"rejected\": \"I think the capital of France might be London or Paris, I'm not entirely sure.\"}", - "language": "python", - "source_html": "{"prompt": [{"role": "user", "content": "What is the capital of France?"}], "chosen": "The capital of France is Paris. It is the largest city in France and serves as the country's political, economic, and cultural center.", "rejected": "I think the capital of France might be London or Paris, I'm not entirely sure."}\n" - }, - { - "type": "markdown", - "source": "#### Tulu3 Preference Dataset Format\nThis format contains complete conversation histories for both the chosen (preferred) and rejected responses.\n\nRequired fields:\n- **`chosen`**: Full conversation with the preferred response (list of message objects, last must be assistant)\n- **`rejected`**: Full conversation with the rejected response (list of message objects, last must be assistant)", - "source_html": "

Tulu3 Preference Dataset Format

\n

This format contains complete conversation histories for both the chosen (preferred) and rejected responses.

\n

Required fields:

\n
    \n
  • chosen: Full conversation with the preferred response (list of message objects, last must be assistant)
  • \n
  • rejected: Full conversation with the rejected response (list of message objects, last must be assistant)
  • \n
\n" - }, - { - "type": "code", - "source": "{\"chosen\": [{\"role\": \"user\", \"content\": \"What is the capital of France?\"}, {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"}], \"rejected\": [{\"role\": \"user\", \"content\": \"What is the capital of France?\"}, {\"role\": \"assistant\", \"content\": \"I'm not sure, but I think it might be London or Paris.\"}]}", - "language": "python", - "source_html": "{"chosen": [{"role": "user", "content": "What is the capital of France?"}, {"role": "assistant", "content": "The capital of France is Paris."}], "rejected": [{"role": "user", "content": "What is the capital of France?"}, {"role": "assistant", "content": "I'm not sure, but I think it might be London or Paris."}]}\n" - }, - { - "type": "markdown", - "source": "#### HelpSteer Dataset Format\nThis format uses numeric preference scores to indicate which response is better. The context can be either a simple string or an array of message objects.\n\nRequired fields:\n- **`context`**: The input context (can be a string or array of message objects)\n- **`response1`**: First response option\n- **`response2`**: Second response option\n- **`overall_preference`**: Preference score where negative values mean response1 is preferred, positive values mean response2 is preferred, and 0 indicates a tie", - "source_html": "

HelpSteer Dataset Format

\n

This format uses numeric preference scores to indicate which response is better. The context can be either a simple string or an array of message objects.

\n

Required fields:

\n
    \n
  • context: The input context (can be a string or array of message objects)
  • \n
  • response1: First response option
  • \n
  • response2: Second response option
  • \n
  • overall_preference: Preference score where negative values mean response1 is preferred, positive values mean response2 is preferred, and 0 indicates a tie
  • \n
\n" - }, - { - "type": "code", - "source": "{\"context\": \"Explain how to use git rebase\", \"response1\": \"Git rebase is a command that rewrites commit history by moving or combining commits. Use 'git rebase main' to reapply your branch commits on top of main. This creates a linear history and avoids merge commits.\", \"response2\": \"Use git rebase to change commits. Just type git rebase and it will work.\", \"overall_preference\": -2}", - "language": "python", - "source_html": "{"context": "Explain how to use git rebase", "response1": "Git rebase is a command that rewrites commit history by moving or combining commits. Use 'git rebase main' to reapply your branch commits on top of main. This creates a linear history and avoids merge commits.", "response2": "Use git rebase to change commits. Just type git rebase and it will work.", "overall_preference": -2}\n" - }, - { - "type": "markdown", - "source": "### 3. Create Dataset FileSet and Upload Training Data", - "source_html": "

3. Create Dataset FileSet and Upload Training Data

\n" - }, - { - "type": "markdown", - "source": "Install huggingface datasets package to download public [nvidia/HelpSteer3](https://huggingface.co/datasets/nvidia/HelpSteer3) dataset if it's not installed in your Python environment:\n\n```sh\npip install datasets\n```", - "source_html": "

Install huggingface datasets package to download public nvidia/HelpSteer3 dataset if it's not installed in your Python environment:

\n
pip install datasets\n
\n" - }, - { - "type": "markdown", - "source": "#### Download nvidia/HelpSteer3 Dataset", - "source_html": "

Download nvidia/HelpSteer3 Dataset

\n" - }, - { - "type": "code", - "source": "from pathlib import Path\nfrom datasets import load_dataset, Dataset\nds = load_dataset(\"nvidia/HelpSteer3\", \"preference\")\n\n# Adjust these values to change the size of the training and validation sets\n# The larger the datasets, the better the model will perform but longer the training will take\n# For the purpose of this tutorial, we'll use a small subset of the dataset\ntraining_size = 3000\nvalidation_size = 300\nDATASET_PATH = Path(\"dpo-dataset\").absolute()\n\n# Get training split and verify it's a Dataset (not IterableDataset)\ntrain_dataset = ds[\"train\"]\nvalidation_dataset = ds[\"validation\"]\nassert isinstance(train_dataset, Dataset), \"Expected Dataset type\"\nassert isinstance(validation_dataset, Dataset), \"Expected Dataset type\"\n\n# Select subsets and save to JSONL files\ntesting_ds = train_dataset.select(range(training_size))\nvalidation_ds = validation_dataset.select(range(validation_size))\n\n# Create directory if it doesn't exist\nos.makedirs(DATASET_PATH, exist_ok=True)\n\n# Save subsets to JSONL files\ntesting_ds.to_json(f\"{DATASET_PATH}/training.jsonl\")\nvalidation_ds.to_json(f\"{DATASET_PATH}/validation.jsonl\")\n\nprint(f\"Saved training.jsonl with {len(testing_ds)} rows\")\nprint(f\"Saved validation.jsonl with {len(validation_ds)} rows\")", - "language": "python", - "source_html": "from pathlib import Path\nfrom datasets import load_dataset, Dataset\nds = load_dataset("nvidia/HelpSteer3", "preference")\n\n# Adjust these values to change the size of the training and validation sets\n# The larger the datasets, the better the model will perform but longer the training will take\n# For the purpose of this tutorial, we'll use a small subset of the dataset\ntraining_size = 3000\nvalidation_size = 300\nDATASET_PATH = Path("dpo-dataset").absolute()\n\n# Get training split and verify it's a Dataset (not IterableDataset)\ntrain_dataset = ds["train"]\nvalidation_dataset = ds["validation"]\nassert isinstance(train_dataset, Dataset), "Expected Dataset type"\nassert isinstance(validation_dataset, Dataset), "Expected Dataset type"\n\n# Select subsets and save to JSONL files\ntesting_ds = train_dataset.select(range(training_size))\nvalidation_ds = validation_dataset.select(range(validation_size))\n\n# Create directory if it doesn't exist\nos.makedirs(DATASET_PATH, exist_ok=True)\n\n# Save subsets to JSONL files\ntesting_ds.to_json(f"{DATASET_PATH}/training.jsonl")\nvalidation_ds.to_json(f"{DATASET_PATH}/validation.jsonl")\n\nprint(f"Saved training.jsonl with {len(testing_ds)} rows")\nprint(f"Saved validation.jsonl with {len(validation_ds)} rows")\n" - }, - { - "type": "markdown", - "source": "#### Upload Training Data", - "source_html": "

Upload Training Data

\n" - }, - { - "type": "code", - "source": "# Create fileset to store DPO training data\nDATASET_NAME = \"dpo-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"dpo training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate training data is uploaded correctly\nprint(\"Training data:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", - "language": "python", - "source_html": "# Create fileset to store DPO training data\nDATASET_NAME = "dpo-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="dpo training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate training data is uploaded correctly\nprint("Training data:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" - }, - { - "type": "markdown", - "source": "### 4. Secrets Setup\n\nIf you plan to use NGC or HuggingFace models, you'll need to configure authentication:\n\n- **NGC models** (`ngc://` URIs): Requires NGC API key\n- **HuggingFace models** (`hf://` URIs): Requires HF token for gated/private models\n\n\nConfigure these as secrets in your platform. See [Managing Secrets](../../get-started/concepts/manage-secrets.md) for detailed instructions.\n\nGet your credentials to access base models:\n- [NGC API Key](https://ngc.nvidia.com/) (Setup → Generate API Key)\n- [HuggingFace Token](https://huggingface.co/settings/tokens) (Create token with Read access)\n\n\n---\n\n#### Quick Setup Example\n\nIn this tutorial we are going to work with [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model from HuggingFace. Ensure that you have sufficient permissions to download the model. If you cannot see the files in the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) Hugging Face page, request access\n\n**HuggingFace Authentication:**\n- For gated models (Llama, Gemma), you must provide a HuggingFace token via the `token_secret` parameter\n- Get your token from [HuggingFace Settings](https://huggingface.co/settings/tokens) (requires Read access)\n- Accept the model's terms on the HuggingFace model page before using it. Example: [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main)\n- For public models, you can omit the `token_secret` parameter when creating a fileset for model in the next step", - "source_html": "

4. Secrets Setup

\n

If you plan to use NGC or HuggingFace models, you'll need to configure authentication:

\n
    \n
  • NGC models (ngc:// URIs): Requires NGC API key
  • \n
  • HuggingFace models (hf:// URIs): Requires HF token for gated/private models
  • \n
\n

Configure these as secrets in your platform. See Managing Secrets for detailed instructions.

\n

Get your credentials to access base models:

\n\n
\n

Quick Setup Example

\n

In this tutorial we are going to work with meta-llama/Llama-3.2-1B-Instruct model from HuggingFace. Ensure that you have sufficient permissions to download the model. If you cannot see the files in the meta-llama/Llama-3.2-1B-Instruct Hugging Face page, request access

\n

HuggingFace Authentication:

\n
    \n
  • For gated models (Llama, Gemma), you must provide a HuggingFace token via the token_secret parameter
  • \n
  • Get your token from HuggingFace Settings (requires Read access)
  • \n
  • Accept the model's terms on the HuggingFace model page before using it. Example: meta-llama/Llama-3.2-1B-Instruct
  • \n
  • For public models, you can omit the token_secret parameter when creating a fileset for model in the next step
  • \n
\n" - }, - { - "type": "code", - "source": "# Export the HF_TOKEN and NGC_API_KEY environment variables if they are not already set\nHF_TOKEN = os.getenv(\"HF_TOKEN\")\nNGC_API_KEY = os.getenv(\"NGC_API_KEY\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n name=name,\n workspace=\"default\",\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name, workspace=\"default\")\n\n\n# Create HuggingFace token secret\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(\"HF_TOKEN secret:\")\nprint(hf_secret.model_dump_json(indent=2))\n\n# Create NGC API key secret\n# Uncomment the line below if you have NGC API Key and want to finetune NGC models\n# ngc_api_key = create_or_get_secret(\"ngc-api-key\", NGC_API_KEY, \"NGC_API_KEY\")", - "language": "python", - "source_html": "# Export the HF_TOKEN and NGC_API_KEY environment variables if they are not already set\nHF_TOKEN = os.getenv("HF_TOKEN")\nNGC_API_KEY = os.getenv("NGC_API_KEY")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f"{label} is not set")\n try:\n secret = client.secrets.create(\n name=name,\n workspace="default",\n value=value,\n )\n print(f"Created secret: {name}")\n return secret\n except ConflictError:\n print(f"Secret '{name}' already exists, continuing...")\n return client.secrets.retrieve(name=name, workspace="default")\n\n\n# Create HuggingFace token secret\nhf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN")\nprint("HF_TOKEN secret:")\nprint(hf_secret.model_dump_json(indent=2))\n\n# Create NGC API key secret\n# Uncomment the line below if you have NGC API Key and want to finetune NGC models\n# ngc_api_key = create_or_get_secret("ngc-api-key", NGC_API_KEY, "NGC_API_KEY")\n" - }, - { - "type": "markdown", - "source": "### 5. Create Base Model FileSet and Model Entity\n\nCreate a fileset pointing to [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model in HuggingFace that we will train with DPO. Then create a Model Entity that references this fileset. Model downloading will take place at the DPO finetuning job creation time.\n\nNote: for public models, you can omit the `token_secret` parameter when creating a model fileset.", - "source_html": "

5. Create Base Model FileSet and Model Entity

\n

Create a fileset pointing to meta-llama/Llama-3.2-1B-Instruct model in HuggingFace that we will train with DPO. Then create a Model Entity that references this fileset. Model downloading will take place at the DPO finetuning job creation time.

\n

Note: for public models, you can omit the token_secret parameter when creating a model fileset.

\n" - }, - { - "type": "code", - "source": "import time\n\n# Create a fileset pointing to the desired HuggingFace model\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = \"meta-llama/Llama-3.2-1B-Instruct\"\nMODEL_NAME = \"llama-3-2-1b-base\"\n\n# Ensure you have a HuggingFace token secret created\ntry:\n base_model_fs = client.files.filesets.create(\n workspace=\"default\",\n name=MODEL_NAME,\n description=\"Llama 3.2 1B base model from HuggingFace\",\n storage=HuggingfaceStorageConfigParam(\n type=\"huggingface\",\n # repo_id is the full model name from Hugging Face\n repo_id=HF_REPO_ID,\n repo_type=\"model\",\n # we use the secret created in the previous step\n token_secret=hf_secret.name\n )\n )\n print(f\"Created base model fileset: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model fileset already exists. Skipping creation.\")\n base_model_fs = client.files.filesets.retrieve(\n workspace=\"default\",\n name=MODEL_NAME,\n )\n\n# Create Model Entity referencing the FileSet\ntry:\n base_model = client.models.create(\n workspace=\"default\",\n name=MODEL_NAME,\n fileset=f\"default/{MODEL_NAME}\",\n )\n print(f\"Created Model Entity: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model already exists. Updating fileset if different.\")\n base_model = client.models.update(\n workspace=\"default\",\n name=MODEL_NAME,\n fileset=f\"default/{MODEL_NAME}\",\n )\n\nprint(f\"\\nBase model fileset: fileset://default/{base_model.name}\")\nprint(\"Base model fileset files list:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=\"default\").data], indent=2))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint(\"\\nWaiting for ModelSpec to be populated...\")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n time.sleep(2)\n base_model = client.models.retrieve(\n workspace=\"default\",\n name=MODEL_NAME,\n )\n\nprint(f\"ModelSpec populated: {base_model.spec}\")", - "language": "python", - "source_html": "import time\n\n# Create a fileset pointing to the desired HuggingFace model\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = "meta-llama/Llama-3.2-1B-Instruct"\nMODEL_NAME = "llama-3-2-1b-base"\n\n# Ensure you have a HuggingFace token secret created\ntry:\n base_model_fs = client.files.filesets.create(\n workspace="default",\n name=MODEL_NAME,\n description="Llama 3.2 1B base model from HuggingFace",\n storage=HuggingfaceStorageConfigParam(\n type="huggingface",\n # repo_id is the full model name from Hugging Face\n repo_id=HF_REPO_ID,\n repo_type="model",\n # we use the secret created in the previous step\n token_secret=hf_secret.name\n )\n )\n print(f"Created base model fileset: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model fileset already exists. Skipping creation.")\n base_model_fs = client.files.filesets.retrieve(\n workspace="default",\n name=MODEL_NAME,\n )\n\n# Create Model Entity referencing the FileSet\ntry:\n base_model = client.models.create(\n workspace="default",\n name=MODEL_NAME,\n fileset=f"default/{MODEL_NAME}",\n )\n print(f"Created Model Entity: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model already exists. Updating fileset if different.")\n base_model = client.models.update(\n workspace="default",\n name=MODEL_NAME,\n fileset=f"default/{MODEL_NAME}",\n )\n\nprint(f"\\nBase model fileset: fileset://default/{base_model.name}")\nprint("Base model fileset files list:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace="default").data], indent=2))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint("\\nWaiting for ModelSpec to be populated...")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds")\n time.sleep(2)\n base_model = client.models.retrieve(\n workspace="default",\n name=MODEL_NAME,\n )\n\nprint(f"ModelSpec populated: {base_model.spec}")\n" - }, - { - "type": "markdown", - "source": "### 6. Create DPO Finetuning Job\nCreate a customization job with an inline target referencing the base model and dataset filesets created in previous steps.", - "source_html": "

6. Create DPO Finetuning Job

\n

Create a customization job with an inline target referencing the base model and dataset filesets created in previous steps.

\n" - }, - { - "type": "markdown", - "source": "**Target `model_uri` Format:**\n\nCurrently, `model_uri` must reference a FileSet:\n- **FileSet:** `fileset://{workspace}/{fileset-name}`\n\nSupport for direct HuggingFace (`hf://`) and NGC (`ngc://`) URIs is coming soon. For now, create a fileset as shown in the previous step, and the HuggingFace model will be downloaded at the beginning the finetuning job.\n\n**GPU Requirements:**\n- 1B models: 1 GPU (24GB+ VRAM)\n- 3B models: 1-2 GPUs \n- 8B models: 2-4 GPUs\n- 70B models: 8+ GPUs \n\nAdjust `num_gpus_per_node` and `tensor_parallel_size` based on your model size.\n\n**Important**\n\nWhen setting `val_check_interval` for DPO, use a fractional value (e.g., `0.5` for twice per epoch) or omit it entirely (validates once at end of epoch). Avoid integer step counts — they may not divide evenly into the total training steps, which can prevent validation from running on the final step.", - "source_html": "

Target model_uri Format:

\n

Currently, model_uri must reference a FileSet:

\n
    \n
  • FileSet: fileset://{workspace}/{fileset-name}
  • \n
\n

Support for direct HuggingFace (hf://) and NGC (ngc://) URIs is coming soon. For now, create a fileset as shown in the previous step, and the HuggingFace model will be downloaded at the beginning the finetuning job.

\n

GPU Requirements:

\n
    \n
  • 1B models: 1 GPU (24GB+ VRAM)
  • \n
  • 3B models: 1-2 GPUs
  • \n
  • 8B models: 2-4 GPUs
  • \n
  • 70B models: 8+ GPUs
  • \n
\n

Adjust num_gpus_per_node and tensor_parallel_size based on your model size.

\n

Important

\n

When setting val_check_interval for DPO, use a fractional value (e.g., 0.5 for twice per epoch) or omit it entirely (validates once at end of epoch). Avoid integer step counts — they may not divide evenly into the total training steps, which can prevent validation from running on the final step.

\n" - }, - { - "type": "code", - "source": "import uuid\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f\"my-dpo-job-{job_suffix}\"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=DpoTrainingParam(\n type=\"dpo\",\n epochs=1,\n batch_size=16,\n learning_rate=0.00005,\n max_seq_length=4096,\n ref_policy_kl_penalty=0.1,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n )\n)\n\nprint(f\"Job ID: {job.name}\")\nprint(f\"Output model: {job.spec.output.name}\")", - "language": "python", - "source_html": "import uuid\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f"my-dpo-job-{job_suffix}"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=DpoTrainingParam(\n type="dpo",\n epochs=1,\n batch_size=16,\n learning_rate=0.00005,\n max_seq_length=4096,\n ref_policy_kl_penalty=0.1,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n )\n)\n\nprint(f"Job ID: {job.name}")\nprint(f"Output model: {job.spec.output.name}")\n" - }, - { - "type": "markdown", - "source": "### 7. Track Training Progress", - "source_html": "

7. Track Training Progress

\n" - }, - { - "type": "code", - "source": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace=\"default\"\n )\n \n clear_output(wait=True)\n print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get(\"step\")\n max_steps = task_details.get(\"max_steps\")\n training_phase = task_details.get(\"phase\")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n if training_phase:\n print(f\"Training Phase: {training_phase}\")\n else:\n print(\"Training step not started yet or progress info not available\")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n break\n \n time.sleep(10)", - "language": "python", - "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace="default"\n )\n \n clear_output(wait=True)\n print(f"Job Status: {status.model_dump_json(indent=2)}")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get("step")\n max_steps = task_details.get("max_steps")\n training_phase = task_details.get("phase")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)")\n if training_phase:\n print(f"Training Phase: {training_phase}")\n else:\n print("Training step not started yet or progress info not available")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n break\n \n time.sleep(10)\n" - }, - { - "type": "markdown", - "source": "**Interpreting DPO Training Metrics:**\n\nDPO training produces several key metrics:\n\n| Metric | Description | What to Look For |\n|--------|-------------|------------------|\n| **loss** | Total training loss (preference_loss + sft_loss) | Should decrease over training |\n| **preference_loss** | Core DPO loss measuring preference learning | Starts near ln(2) ≈ 0.693, should decrease |\n| **sft_loss** | SFT regularization term (often 0 for pure DPO) | Depends on configuration |\n| **accuracy** | Fraction of samples where chosen > rejected | Should increase toward 80-95%+ |\n| **rewards_chosen_mean** | Average implicit reward for chosen responses | Should be positive |\n| **rewards_rejected_mean** | Average implicit reward for rejected responses | Should be negative |\n\n**Key Indicators:**\n\n- **Reward Margin** = `rewards_chosen_mean - rewards_rejected_mean`\n - Should be positive and increasing\n - Indicates the model is learning to distinguish preferences\n\n- **Accuracy Interpretation:**\n - 50% = random chance (no learning)\n - 66-75% = early/moderate learning\n - 80%+ = good preference learning\n - 95%+ = strong preference alignment\n\n**Troubleshooting:**\n\n- **Loss near ln(2) ≈ 0.693**: Model is at random chance level, training just starting or not learning\n- **Accuracy stuck at ~50%**: Check data quality, increase learning rate, or verify preference labels\n- **Negative reward margin**: Model is learning the wrong direction—check chosen/rejected labels\n- **Loss increasing**: Learning rate too high or data quality issues\n\n**Note:** Training metrics measure optimization progress, not final model quality. Always evaluate the deployed model on your specific use case.", - "source_html": "

Interpreting DPO Training Metrics:

\n

DPO training produces several key metrics:

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
MetricDescriptionWhat to Look For
lossTotal training loss (preference_loss + sft_loss)Should decrease over training
preference_lossCore DPO loss measuring preference learningStarts near ln(2) ≈ 0.693, should decrease
sft_lossSFT regularization term (often 0 for pure DPO)Depends on configuration
accuracyFraction of samples where chosen > rejectedShould increase toward 80-95%+
rewards_chosen_meanAverage implicit reward for chosen responsesShould be positive
rewards_rejected_meanAverage implicit reward for rejected responsesShould be negative
\n

Key Indicators:

\n
    \n
  • \n

    Reward Margin = rewards_chosen_mean - rewards_rejected_mean

    \n
      \n
    • Should be positive and increasing
    • \n
    • Indicates the model is learning to distinguish preferences
    • \n
    \n
  • \n
  • \n

    Accuracy Interpretation:

    \n
      \n
    • 50% = random chance (no learning)
    • \n
    • 66-75% = early/moderate learning
    • \n
    • 80%+ = good preference learning
    • \n
    • 95%+ = strong preference alignment
    • \n
    \n
  • \n
\n

Troubleshooting:

\n
    \n
  • Loss near ln(2) ≈ 0.693: Model is at random chance level, training just starting or not learning
  • \n
  • Accuracy stuck at ~50%: Check data quality, increase learning rate, or verify preference labels
  • \n
  • Negative reward margin: Model is learning the wrong direction—check chosen/rejected labels
  • \n
  • Loss increasing: Learning rate too high or data quality issues
  • \n
\n

Note: Training metrics measure optimization progress, not final model quality. Always evaluate the deployed model on your specific use case.

\n" - }, - { - "type": "markdown", - "source": "### 8. Deploy Fine-Tuned Model\n\nOnce training completes, deploy using the Deployment Management Service:", - "source_html": "

8. Deploy Fine-Tuned Model

\n

Once training completes, deploy using the Deployment Management Service:

\n" - }, - { - "type": "code", - "source": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))", - "language": "python", - "source_html": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))\n" - }, - { - "type": "code", - "source": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"dpo-model-deployment-cfg-{deploy_suffix}\"\nDEPLOYMENT_NAME = f\"dpo-model-deployment-{deploy_suffix}\"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n image_tag=\"1.13.1\",\n gpu=1,\n model_name=job.spec.output.name, # ModelEntity name from training,\n model_namespace=\"default\", # Workspace where ModelEntity lives\n )\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace=\"default\"\n)\n\nprint(f\"Deployment name: {deployment.name}\")\nprint(f\"Deployment status: {deployment_status.status}\")", - "language": "python", - "source_html": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"dpo-model-deployment-cfg-{deploy_suffix}"\nDEPLOYMENT_NAME = f"dpo-model-deployment-{deploy_suffix}"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name="nvcr.io/nim/nvidia/llm-nim",\n image_tag="1.13.1",\n gpu=1,\n model_name=job.spec.output.name, # ModelEntity name from training,\n model_namespace="default", # Workspace where ModelEntity lives\n )\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace="default"\n)\n\nprint(f"Deployment name: {deployment.name}")\nprint(f"Deployment status: {deployment_status.status}")\n" - }, - { - "type": "markdown", - "source": "### Monitor status of deployment", - "source_html": "

Monitor status of deployment

\n" - }, - { - "type": "code", - "source": "import time\nfrom IPython.display import clear_output\n\n# Poll deployment status every 15 seconds until ready\nTIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f\"Monitoring deployment '{deployment.name}'...\")\nprint(f\"Timeout: {TIMEOUT_MINUTES} minutes\\n\")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace=\"default\"\n )\n \n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n \n clear_output(wait=True)\n print(f\"Deployment: {deployment.name}\")\n print(f\"Status: {deployment_status.status}\")\n print(f\"Elapsed time: {elapsed_min}m {elapsed_sec}s\")\n \n # Check if deployment is ready\n if deployment_status.status == \"READY\":\n print(\"\\nDeployment is ready!\")\n if not client.models.wait_for_gateway(deployment.name, workspace=\"default\", timeout=60):\n raise RuntimeError(\"Inference gateway did not become ready\")\n break\n \n # Check for failure states\n if deployment_status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n raise RuntimeError(f\"Deployment failed with status: {deployment_status.status}\")\n \n # Check timeout\n if elapsed > timeout_seconds:\n raise TimeoutError(f\"Deployment timeout after {TIMEOUT_MINUTES} minutes\")\n \n time.sleep(15)", - "language": "python", - "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll deployment status every 15 seconds until ready\nTIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f"Monitoring deployment '{deployment.name}'...")\nprint(f"Timeout: {TIMEOUT_MINUTES} minutes\\n")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace="default"\n )\n \n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n \n clear_output(wait=True)\n print(f"Deployment: {deployment.name}")\n print(f"Status: {deployment_status.status}")\n print(f"Elapsed time: {elapsed_min}m {elapsed_sec}s")\n \n # Check if deployment is ready\n if deployment_status.status == "READY":\n print("\\nDeployment is ready!")\n if not client.models.wait_for_gateway(deployment.name, workspace="default", timeout=60):\n raise RuntimeError("Inference gateway did not become ready")\n break\n \n # Check for failure states\n if deployment_status.status in ("FAILED", "ERROR", "TERMINATED", "LOST"):\n raise RuntimeError(f"Deployment failed with status: {deployment_status.status}")\n \n # Check timeout\n if elapsed > timeout_seconds:\n raise TimeoutError(f"Deployment timeout after {TIMEOUT_MINUTES} minutes")\n \n time.sleep(15)\n" - }, - { - "type": "markdown", - "source": "The deployment service automatically:\n- Downloads model weights from the Files service\n- Provisions storage (PVC) for the weights\n- Configures and starts the NIM container\n\n**Multi-GPU Deployment:**\n\nFor larger models requiring multiple GPUs, configure parallelism with environment variables:\n\n```python\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=\"sft-model-config-multigpu\",\n \n nim_deployment={\n \"image_name\": \"nvcr.io/nim/nvidia/llm-nim\",\n \"image_tag\": \"1.13.1\",\n \"gpu\": 2, # Total GPUs\n \"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"2\", # Tensor parallelism\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"1\" # Pipeline parallelism\n }\n }\n)\n```", - "source_html": "

The deployment service automatically:

\n
    \n
  • Downloads model weights from the Files service
  • \n
  • Provisions storage (PVC) for the weights
  • \n
  • Configures and starts the NIM container
  • \n
\n

Multi-GPU Deployment:

\n

For larger models requiring multiple GPUs, configure parallelism with environment variables:

\n
deployment_config = client.inference.deployment_configs.create(\n    workspace="default",\n    name="sft-model-config-multigpu",\n    \n    nim_deployment={\n        "image_name": "nvcr.io/nim/nvidia/llm-nim",\n        "image_tag": "1.13.1",\n        "gpu": 2,  # Total GPUs\n        "additional_envs": {\n            "NIM_TENSOR_PARALLEL_SIZE": "2",  # Tensor parallelism\n            "NIM_PIPELINE_PARALLEL_SIZE": "1"  # Pipeline parallelism\n        }\n    }\n)\n
\n" - }, - { - "type": "markdown", - "source": "**Single-Node Constraint:** Model deployments are limited to a single node. The maximum `gpu` value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.\n\n---\n\n#### GPU Parallelism\n\nBy default, NIM uses all GPUs for tensor parallelism (TP). You can customize this behavior using the `NIM_TENSOR_PARALLEL_SIZE` and `NIM_PIPELINE_PARALLEL_SIZE` environment variables.\n\n| Strategy | Description | Best For |\n|----------|-------------|----------|\n| **Tensor Parallel (TP)** | Splits model layers across GPUs | Lowest latency |\n| **Pipeline Parallel (PP)** | Splits model depth across GPUs | Highest throughput |\n\n**Formula:** `gpu` = `NIM_TENSOR_PARALLEL_SIZE` × `NIM_PIPELINE_PARALLEL_SIZE`\n\n---\n\n#### Example Configurations\n\n**Default (TP=8, PP=1) — Lowest Latency**\n```\n\"gpu\": 8\n# NIM automatically sets NIM_TENSOR_PARALLEL_SIZE=8\n```\n\n**Balanced (TP=4, PP=2)**\n```\n\"gpu\": 8,\n\"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"4\",\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"2\"\n}\n```\n\n**Throughput Optimized (TP=2, PP=4)**\n```\n\"gpu\": 8,\n\"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"2\",\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"4\"\n}\n```", - "source_html": "

Single-Node Constraint: Model deployments are limited to a single node. The maximum gpu value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.

\n
\n

GPU Parallelism

\n

By default, NIM uses all GPUs for tensor parallelism (TP). You can customize this behavior using the NIM_TENSOR_PARALLEL_SIZE and NIM_PIPELINE_PARALLEL_SIZE environment variables.

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
StrategyDescriptionBest For
Tensor Parallel (TP)Splits model layers across GPUsLowest latency
Pipeline Parallel (PP)Splits model depth across GPUsHighest throughput
\n

Formula: gpu = NIM_TENSOR_PARALLEL_SIZE × NIM_PIPELINE_PARALLEL_SIZE

\n
\n

Example Configurations

\n

Default (TP=8, PP=1) — Lowest Latency

\n
"gpu": 8\n# NIM automatically sets NIM_TENSOR_PARALLEL_SIZE=8\n
\n

Balanced (TP=4, PP=2)

\n
"gpu": 8,\n"additional_envs": {\n    "NIM_TENSOR_PARALLEL_SIZE": "4",\n    "NIM_PIPELINE_PARALLEL_SIZE": "2"\n}\n
\n

Throughput Optimized (TP=2, PP=4)

\n
"gpu": 8,\n"additional_envs": {\n    "NIM_TENSOR_PARALLEL_SIZE": "2",\n    "NIM_PIPELINE_PARALLEL_SIZE": "4"\n}\n
\n" - }, - { - "type": "markdown", - "source": "### 9. Evaluate Your Model\n\nAfter training, evaluate whether your model meets your requirements:\n\n#### Quick Manual Evaluation", - "source_html": "

9. Evaluate Your Model

\n

After training, evaluate whether your model meets your requirements:

\n

Quick Manual Evaluation

\n" - }, - { - "type": "code", - "source": "# Wait for deployment to be ready, then test\nmessages = [\n {\"role\": \"system\", \"content\": \"You are a helpful assistant.\"},\n {\"role\": \"user\", \"content\": \"Write a short email to my colleague.\"}\n]\n\nresponse = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": f\"default/{job.spec.output.name}\", # Match the model_name from deployment config\n \"messages\": messages,\n \"temperature\": 0.7,\n \"max_tokens\": 256\n }\n)\n\n# Display prompt and completion\nprint(\"=\" * 60)\nprint(\"PROMPT\")\nprint(\"=\" * 60)\nfor msg in messages:\n print(f\"[{msg['role'].upper()}]\")\n print(msg[\"content\"])\n print()\n\nprint(\"=\" * 60)\nprint(\"COMPLETION\")\nprint(\"=\" * 60)\nprint(\"[ASSISTANT]\")\ncompletion = response[\"choices\"][0][\"message\"][\"content\"]\nprint(completion)", - "language": "python", - "source_html": "# Wait for deployment to be ready, then test\nmessages = [\n {"role": "system", "content": "You are a helpful assistant."},\n {"role": "user", "content": "Write a short email to my colleague."}\n]\n\nresponse = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment.name,\n workspace="default",\n body={\n "model": f"default/{job.spec.output.name}", # Match the model_name from deployment config\n "messages": messages,\n "temperature": 0.7,\n "max_tokens": 256\n }\n)\n\n# Display prompt and completion\nprint("=" * 60)\nprint("PROMPT")\nprint("=" * 60)\nfor msg in messages:\n print(f"[{msg['role'].upper()}]")\n print(msg["content"])\n print()\n\nprint("=" * 60)\nprint("COMPLETION")\nprint("=" * 60)\nprint("[ASSISTANT]")\ncompletion = response["choices"][0]["message"]["content"]\nprint(completion)\n" - }, - { - "type": "markdown", - "source": "#### Evaluation Best Practices\n\n**Manual Evaluation** (Recommended)\n- Test with real-world examples from your use case\n- Compare responses to base model and expected outputs\n- Verify the model exhibits desired behavior changes\n- Check edge cases and error handling\n\n**What to look for:**\n- ✅ Model follows your desired output format\n- ✅ Applies domain knowledge correctly\n- ✅ Maintains general language capabilities\n- ✅ Avoids unwanted behaviors or biases\n- ❌ Doesn't hallucinate facts not in training data\n- ❌ Doesn't produce repetitive or nonsensical outputs\n\n---\n\n## Hyperparameters\n\nFor detailed information on all available hyperparameters, recommended values, and tuning guidance, see the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n\n---\n\n\n## Troubleshooting\n\n**Job fails during model download:**\n- Verify authentication secrets are configured (see [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n- Check the `model_uri` format is correct (`fileset://`)\n- Ensure you have accepted the model's terms of service on HuggingFace\n- Check job status and logs: `client.customization.jobs.retrieve(name=job.name, workspace=\"default\")`\n\n**Job fails with OOM (Out of Memory) error:**\n1. **First try:** Reduce `micro_batch_size` from 2 to 1\n2. **Still OOM:** Reduce `batch_size` from 16 to 8\n3. **Still OOM:** Reduce `max_seq_length` from 2048 to 1024 or 512\n4. **Last resort:** Increase GPU count and use `tensor_parallel_size` for model sharding\n\n**Loss curves not decreasing (underfitting):**\n- Increase training duration: `epochs: 5-10` instead of 3\n- Adjust learning rate: Try `1e-5` to `1e-4`\n- Add warmup: Set `warmup_steps` to ~10% of total training steps\n- Check data quality: Verify formatting, remove duplicates, ensure diversity\n\n**Training loss decreases but validation loss increases (overfitting):**\n- Reduce epochs: Try `epochs: 1-2` instead of 5+\n- Lower learning rate: Use `2e-5` or `1e-5`\n- Increase dataset size and diversity\n- Verify train/validation split has no data leakage\n\n**Model output quality is poor despite good training metrics:**\n- Training metrics optimize for loss, not your actual task—evaluate on real use cases\n- Review data quality, format, and diversity—metrics can be misleading with poor data\n- Try a different base model size or architecture\n- Adjust learning rate and batch size\n- Compare to baseline: Test base model to ensure fine-tuning improved performance\n\n**Deployment fails:**\n- Verify output model exists: `client.models.retrieve(name=job.spec.output.name, workspace=\"default\")`\n- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n- Ensure sufficient GPU resources available for model size\n- Verify NIM image tag `1.13.1` is compatible with your model\n\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning", - "source_html": "

Evaluation Best Practices

\n

Manual Evaluation (Recommended)

\n
    \n
  • Test with real-world examples from your use case
  • \n
  • Compare responses to base model and expected outputs
  • \n
  • Verify the model exhibits desired behavior changes
  • \n
  • Check edge cases and error handling
  • \n
\n

What to look for:

\n
    \n
  • ✅ Model follows your desired output format
  • \n
  • ✅ Applies domain knowledge correctly
  • \n
  • ✅ Maintains general language capabilities
  • \n
  • ✅ Avoids unwanted behaviors or biases
  • \n
  • ❌ Doesn't hallucinate facts not in training data
  • \n
  • ❌ Doesn't produce repetitive or nonsensical outputs
  • \n
\n
\n

Hyperparameters

\n

For detailed information on all available hyperparameters, recommended values, and tuning guidance, see the Hyperparameter Reference.

\n
\n

Troubleshooting

\n

Job fails during model download:

\n
    \n
  • Verify authentication secrets are configured (see Managing Secrets)
  • \n
  • For gated HuggingFace models (Llama, Gemma), accept the license on the model page
  • \n
  • Check the model_uri format is correct (fileset://)
  • \n
  • Ensure you have accepted the model's terms of service on HuggingFace
  • \n
  • Check job status and logs: client.customization.jobs.retrieve(name=job.name, workspace="default")
  • \n
\n

Job fails with OOM (Out of Memory) error:

\n
    \n
  1. First try: Reduce micro_batch_size from 2 to 1
  2. \n
  3. Still OOM: Reduce batch_size from 16 to 8
  4. \n
  5. Still OOM: Reduce max_seq_length from 2048 to 1024 or 512
  6. \n
  7. Last resort: Increase GPU count and use tensor_parallel_size for model sharding
  8. \n
\n

Loss curves not decreasing (underfitting):

\n
    \n
  • Increase training duration: epochs: 5-10 instead of 3
  • \n
  • Adjust learning rate: Try 1e-5 to 1e-4
  • \n
  • Add warmup: Set warmup_steps to ~10% of total training steps
  • \n
  • Check data quality: Verify formatting, remove duplicates, ensure diversity
  • \n
\n

Training loss decreases but validation loss increases (overfitting):

\n
    \n
  • Reduce epochs: Try epochs: 1-2 instead of 5+
  • \n
  • Lower learning rate: Use 2e-5 or 1e-5
  • \n
  • Increase dataset size and diversity
  • \n
  • Verify train/validation split has no data leakage
  • \n
\n

Model output quality is poor despite good training metrics:

\n
    \n
  • Training metrics optimize for loss, not your actual task—evaluate on real use cases
  • \n
  • Review data quality, format, and diversity—metrics can be misleading with poor data
  • \n
  • Try a different base model size or architecture
  • \n
  • Adjust learning rate and batch size
  • \n
  • Compare to baseline: Test base model to ensure fine-tuning improved performance
  • \n
\n

Deployment fails:

\n
    \n
  • Verify output model exists: client.models.retrieve(name=job.spec.output.name, workspace="default")
  • \n
  • Check deployment logs: client.inference.deployments.get_logs(name=deployment.name, workspace="default")
  • \n
  • Ensure sufficient GPU resources available for model size
  • \n
  • Verify NIM image tag 1.13.1 is compatible with your model
  • \n
\n

Next Steps

\n\n" - } - ] -} \ No newline at end of file diff --git a/docs/fern/components/notebooks/dpo-customization-job.ts b/docs/fern/components/notebooks/dpo-customization-job.ts deleted file mode 100644 index 4db1ed71d9..0000000000 --- a/docs/fern/components/notebooks/dpo-customization-job.ts +++ /dev/null @@ -1,205 +0,0 @@ -// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -// SPDX-License-Identifier: Apache-2.0 - -/** Auto-generated by ipynb-to-fern-json.py - do not edit */ -export default { cells: [ - { - "type": "markdown", - "source": "\n\n\n# DPO Customization\n\nLearn how to use the NeMo Platform to create a DPO (Direct Preference Optimization) job using a custom dataset.\n\n## About\n\nDPO is an advanced fine-tuning technique for preference-based alignment. If you're new to fine-tuning, consider starting with [LoRA](./lora-customization-job) or [Full SFT](./sft-customization-job) tutorials first.\n\nDirect Preference Optimization (DPO) is an RL-free alignment algorithm that operates on preference data. Given a prompt and a pair of chosen and rejected responses, DPO aims to increase the probability of the chosen response and decrease the probability of the rejected response relative to a frozen reference model. The actor is initialized using the reference model. For more details, refer to the [DPO paper](https://arxiv.org/pdf/2305.18290).\n\nDPO shares similarities with Full SFT training workflows but differs in a few key ways:\n\n| Aspect | SFT (Supervised Fine-Tuning) | DPO (Direct Preference Optimization) |\n| --- | --- | --- |\n| Data Requirements | Labeled instruction-response pairs where the desired output is explicitly provided | Pairwise preference data, where for a given input, one response is explicitly preferred over another |\n| Learning Objective | Directly teaches the model to generate a specific \"correct\" response | Directly optimizes the model to align with human preferences by maximizing the probability of preferred responses and minimizing rejected ones, without needing an explicit reward model |\n| Alignment Focus | Aligns the model with the specific examples present in its training data | Aligns the model with broader human preferences, which can be more effective for subjective tasks or those without a single \"correct\" answer |\n| Computational Efficiency | Standard fine-tuning efficiency | More computationally efficient than SFT (especially when compared to full RLHF methods) as it bypasses the need to train a separate reward model |\n\n**What you can achieve with DPO:**\n- **Align with human preferences**: Directly optimize your model to produce outputs that align with subjective human preferences without requiring explicit reward modeling\n- **Refine response quality**: Improve helpfulness, harmlessness, honesty, and other nuanced qualities that are easier to compare than to define\n- **Control tone and style**: Adjust the model's communication style, verbosity, formality, and other subjective characteristics\n- **Implement safety guardrails**: Teach the model to avoid harmful or undesirable responses by training on preferred vs. rejected response pairs\n- **Optimize subjective tasks**: Excel at tasks where there are multiple acceptable answers but clear preferences exist (creative writing, dialogue, explanations)\n\n**When to choose DPO:**\n- **Subjective quality matters**: Your task involves style, tone, or other qualities where there's no single \"correct\" answer but clear preferences exist\n- **You have preference data**: You can collect pairwise comparisons (preferred vs. rejected responses) more easily than perfect labeled examples\n- **Refining existing capabilities**: You want to make targeted improvements to an already-trained model without major capability changes\n- **Complex evaluation**: Humans find it easier to compare which of two responses is better than to create the ideal response themselves (especially for multi-turn conversations, creative tasks, or nuanced outputs)\n- **Robust behavior changes**: You need more reliable behavior modification than prompting can provide, without the complexity of full RLHF\n- **Lower compute than RLHF**: You want human preference alignment but with simpler training that doesn't require reinforcement learning infrastructure\n\n**When to choose SFT:**\n- **Clear correct answers**: Your task has objectively correct outputs (code generation, structured data extraction, following specific formats)\n- **High-quality examples**: You have well-labeled input-output pairs that demonstrate exactly what the model should produce\n- **Imitation learning**: You want the model to closely mimic a specific style, format, or knowledge base from expert demonstrations\n- **Foundational capabilities**: You're establishing new task-specific capabilities before fine-tuning preferences (SFT is often done before DPO)\n- **Stable, predictable outputs**: You need consistent formatting or structure that's well-defined in your training examples\n- **Traditional NLP tasks**: Instruction following, translation, summarization, or classification where gold-standard labels exist", - "source_html": "\n\n

DPO Customization

\n

Learn how to use the NeMo Platform to create a DPO (Direct Preference Optimization) job using a custom dataset.

\n

About

\n

DPO is an advanced fine-tuning technique for preference-based alignment. If you're new to fine-tuning, consider starting with LoRA or Full SFT tutorials first.

\n

Direct Preference Optimization (DPO) is an RL-free alignment algorithm that operates on preference data. Given a prompt and a pair of chosen and rejected responses, DPO aims to increase the probability of the chosen response and decrease the probability of the rejected response relative to a frozen reference model. The actor is initialized using the reference model. For more details, refer to the DPO paper.

\n

DPO shares similarities with Full SFT training workflows but differs in a few key ways:

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
AspectSFT (Supervised Fine-Tuning)DPO (Direct Preference Optimization)
Data RequirementsLabeled instruction-response pairs where the desired output is explicitly providedPairwise preference data, where for a given input, one response is explicitly preferred over another
Learning ObjectiveDirectly teaches the model to generate a specific "correct" responseDirectly optimizes the model to align with human preferences by maximizing the probability of preferred responses and minimizing rejected ones, without needing an explicit reward model
Alignment FocusAligns the model with the specific examples present in its training dataAligns the model with broader human preferences, which can be more effective for subjective tasks or those without a single "correct" answer
Computational EfficiencyStandard fine-tuning efficiencyMore computationally efficient than SFT (especially when compared to full RLHF methods) as it bypasses the need to train a separate reward model
\n

What you can achieve with DPO:

\n
    \n
  • Align with human preferences: Directly optimize your model to produce outputs that align with subjective human preferences without requiring explicit reward modeling
  • \n
  • Refine response quality: Improve helpfulness, harmlessness, honesty, and other nuanced qualities that are easier to compare than to define
  • \n
  • Control tone and style: Adjust the model's communication style, verbosity, formality, and other subjective characteristics
  • \n
  • Implement safety guardrails: Teach the model to avoid harmful or undesirable responses by training on preferred vs. rejected response pairs
  • \n
  • Optimize subjective tasks: Excel at tasks where there are multiple acceptable answers but clear preferences exist (creative writing, dialogue, explanations)
  • \n
\n

When to choose DPO:

\n
    \n
  • Subjective quality matters: Your task involves style, tone, or other qualities where there's no single "correct" answer but clear preferences exist
  • \n
  • You have preference data: You can collect pairwise comparisons (preferred vs. rejected responses) more easily than perfect labeled examples
  • \n
  • Refining existing capabilities: You want to make targeted improvements to an already-trained model without major capability changes
  • \n
  • Complex evaluation: Humans find it easier to compare which of two responses is better than to create the ideal response themselves (especially for multi-turn conversations, creative tasks, or nuanced outputs)
  • \n
  • Robust behavior changes: You need more reliable behavior modification than prompting can provide, without the complexity of full RLHF
  • \n
  • Lower compute than RLHF: You want human preference alignment but with simpler training that doesn't require reinforcement learning infrastructure
  • \n
\n

When to choose SFT:

\n
    \n
  • Clear correct answers: Your task has objectively correct outputs (code generation, structured data extraction, following specific formats)
  • \n
  • High-quality examples: You have well-labeled input-output pairs that demonstrate exactly what the model should produce
  • \n
  • Imitation learning: You want the model to closely mimic a specific style, format, or knowledge base from expert demonstrations
  • \n
  • Foundational capabilities: You're establishing new task-specific capabilities before fine-tuning preferences (SFT is often done before DPO)
  • \n
  • Stable, predictable outputs: You need consistent formatting or structure that's well-defined in your training examples
  • \n
  • Traditional NLP tasks: Instruction following, translation, summarization, or classification where gold-standard labels exist
  • \n
\n" - }, - { - "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
\n" - }, - { - "type": "markdown", - "source": "## Quick Start\n\n### 1. Initialize SDK\n\nThe SDK needs to know your NeMo Platform server URL. By default, `http://localhost:8080` is used in accordance with the [Quickstart](../../get-started/quickstart.md) guide. If NeMo Platform is running at a custom location, you can override the URL by setting the `NMP_BASE_URL` environment variable:\n\n```sh\nexport NMP_BASE_URL=\n```", - "source_html": "

Quick Start

\n

1. Initialize SDK

\n

The SDK needs to know your NeMo Platform server URL. By default, http://localhost:8080 is used in accordance with the Quickstart guide. If NeMo Platform is running at a custom location, you can override the URL by setting the NMP_BASE_URL environment variable:

\n
export NMP_BASE_URL=<YOUR_NMP_BASE_URL>\n
\n" - }, - { - "type": "code", - "source": "import json\nimport os\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n DpoTrainingParam,\n ParallelismParamsParam,\n)\n\nNMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace=\"default\"\n)", - "language": "python", - "source_html": "import json\nimport os\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n DpoTrainingParam,\n ParallelismParamsParam,\n)\n\nNMP_BASE_URL = os.environ.get("NMP_BASE_URL", "http://localhost:8080")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace="default"\n)\n" - }, - { - "type": "markdown", - "source": "### 2. Prepare Dataset\n\nCreate your data in JSONL format - one JSON object per line. The platform auto-detects your data format. Supported dataset formats are listed below.\n\n**Flexible Data Setup:**\n- **No validation file?** The platform automatically creates a 10% validation split\n- **Multiple files?** Upload to `training/` or `validation/` subdirectories—they'll be automatically merged\n- **Format detection:** Your data format is auto-detected at training time\n\nIn this tutorial the following dataset directory structure will be used:\n```\nmy_dataset\n`-- training.jsonl\n`-- validation.jsonl\n```", - "source_html": "

2. Prepare Dataset

\n

Create your data in JSONL format - one JSON object per line. The platform auto-detects your data format. Supported dataset formats are listed below.

\n

Flexible Data Setup:

\n
    \n
  • No validation file? The platform automatically creates a 10% validation split
  • \n
  • Multiple files? Upload to training/ or validation/ subdirectories—they'll be automatically merged
  • \n
  • Format detection: Your data format is auto-detected at training time
  • \n
\n

In this tutorial the following dataset directory structure will be used:

\n
my_dataset\n`-- training.jsonl\n`-- validation.jsonl\n
\n" - }, - { - "type": "markdown", - "source": "#### Binary Preference Format\nDPO training requires preference pairs with three fields:\n- **`prompt`**: The input prompt (can be a string or array of message objects)\n- **`chosen`**: The preferred response\n- **`rejected`**: The less preferred response", - "source_html": "

Binary Preference Format

\n

DPO training requires preference pairs with three fields:

\n
    \n
  • prompt: The input prompt (can be a string or array of message objects)
  • \n
  • chosen: The preferred response
  • \n
  • rejected: The less preferred response
  • \n
\n" - }, - { - "type": "code", - "source": "{\"prompt\": [{\"role\": \"user\", \"content\": \"What is the capital of France?\"}], \"chosen\": \"The capital of France is Paris. It is the largest city in France and serves as the country's political, economic, and cultural center.\", \"rejected\": \"I think the capital of France might be London or Paris, I'm not entirely sure.\"}", - "language": "python", - "source_html": "{"prompt": [{"role": "user", "content": "What is the capital of France?"}], "chosen": "The capital of France is Paris. It is the largest city in France and serves as the country's political, economic, and cultural center.", "rejected": "I think the capital of France might be London or Paris, I'm not entirely sure."}\n" - }, - { - "type": "markdown", - "source": "#### Tulu3 Preference Dataset Format\nThis format contains complete conversation histories for both the chosen (preferred) and rejected responses.\n\nRequired fields:\n- **`chosen`**: Full conversation with the preferred response (list of message objects, last must be assistant)\n- **`rejected`**: Full conversation with the rejected response (list of message objects, last must be assistant)", - "source_html": "

Tulu3 Preference Dataset Format

\n

This format contains complete conversation histories for both the chosen (preferred) and rejected responses.

\n

Required fields:

\n
    \n
  • chosen: Full conversation with the preferred response (list of message objects, last must be assistant)
  • \n
  • rejected: Full conversation with the rejected response (list of message objects, last must be assistant)
  • \n
\n" - }, - { - "type": "code", - "source": "{\"chosen\": [{\"role\": \"user\", \"content\": \"What is the capital of France?\"}, {\"role\": \"assistant\", \"content\": \"The capital of France is Paris.\"}], \"rejected\": [{\"role\": \"user\", \"content\": \"What is the capital of France?\"}, {\"role\": \"assistant\", \"content\": \"I'm not sure, but I think it might be London or Paris.\"}]}", - "language": "python", - "source_html": "{"chosen": [{"role": "user", "content": "What is the capital of France?"}, {"role": "assistant", "content": "The capital of France is Paris."}], "rejected": [{"role": "user", "content": "What is the capital of France?"}, {"role": "assistant", "content": "I'm not sure, but I think it might be London or Paris."}]}\n" - }, - { - "type": "markdown", - "source": "#### HelpSteer Dataset Format\nThis format uses numeric preference scores to indicate which response is better. The context can be either a simple string or an array of message objects.\n\nRequired fields:\n- **`context`**: The input context (can be a string or array of message objects)\n- **`response1`**: First response option\n- **`response2`**: Second response option\n- **`overall_preference`**: Preference score where negative values mean response1 is preferred, positive values mean response2 is preferred, and 0 indicates a tie", - "source_html": "

HelpSteer Dataset Format

\n

This format uses numeric preference scores to indicate which response is better. The context can be either a simple string or an array of message objects.

\n

Required fields:

\n
    \n
  • context: The input context (can be a string or array of message objects)
  • \n
  • response1: First response option
  • \n
  • response2: Second response option
  • \n
  • overall_preference: Preference score where negative values mean response1 is preferred, positive values mean response2 is preferred, and 0 indicates a tie
  • \n
\n" - }, - { - "type": "code", - "source": "{\"context\": \"Explain how to use git rebase\", \"response1\": \"Git rebase is a command that rewrites commit history by moving or combining commits. Use 'git rebase main' to reapply your branch commits on top of main. This creates a linear history and avoids merge commits.\", \"response2\": \"Use git rebase to change commits. Just type git rebase and it will work.\", \"overall_preference\": -2}", - "language": "python", - "source_html": "{"context": "Explain how to use git rebase", "response1": "Git rebase is a command that rewrites commit history by moving or combining commits. Use 'git rebase main' to reapply your branch commits on top of main. This creates a linear history and avoids merge commits.", "response2": "Use git rebase to change commits. Just type git rebase and it will work.", "overall_preference": -2}\n" - }, - { - "type": "markdown", - "source": "### 3. Create Dataset FileSet and Upload Training Data", - "source_html": "

3. Create Dataset FileSet and Upload Training Data

\n" - }, - { - "type": "markdown", - "source": "Install huggingface datasets package to download public [nvidia/HelpSteer3](https://huggingface.co/datasets/nvidia/HelpSteer3) dataset if it's not installed in your Python environment:\n\n```sh\npip install datasets\n```", - "source_html": "

Install huggingface datasets package to download public nvidia/HelpSteer3 dataset if it's not installed in your Python environment:

\n
pip install datasets\n
\n" - }, - { - "type": "markdown", - "source": "#### Download nvidia/HelpSteer3 Dataset", - "source_html": "

Download nvidia/HelpSteer3 Dataset

\n" - }, - { - "type": "code", - "source": "from pathlib import Path\nfrom datasets import load_dataset, Dataset\nds = load_dataset(\"nvidia/HelpSteer3\", \"preference\")\n\n# Adjust these values to change the size of the training and validation sets\n# The larger the datasets, the better the model will perform but longer the training will take\n# For the purpose of this tutorial, we'll use a small subset of the dataset\ntraining_size = 3000\nvalidation_size = 300\nDATASET_PATH = Path(\"dpo-dataset\").absolute()\n\n# Get training split and verify it's a Dataset (not IterableDataset)\ntrain_dataset = ds[\"train\"]\nvalidation_dataset = ds[\"validation\"]\nassert isinstance(train_dataset, Dataset), \"Expected Dataset type\"\nassert isinstance(validation_dataset, Dataset), \"Expected Dataset type\"\n\n# Select subsets and save to JSONL files\ntesting_ds = train_dataset.select(range(training_size))\nvalidation_ds = validation_dataset.select(range(validation_size))\n\n# Create directory if it doesn't exist\nos.makedirs(DATASET_PATH, exist_ok=True)\n\n# Save subsets to JSONL files\ntesting_ds.to_json(f\"{DATASET_PATH}/training.jsonl\")\nvalidation_ds.to_json(f\"{DATASET_PATH}/validation.jsonl\")\n\nprint(f\"Saved training.jsonl with {len(testing_ds)} rows\")\nprint(f\"Saved validation.jsonl with {len(validation_ds)} rows\")", - "language": "python", - "source_html": "from pathlib import Path\nfrom datasets import load_dataset, Dataset\nds = load_dataset("nvidia/HelpSteer3", "preference")\n\n# Adjust these values to change the size of the training and validation sets\n# The larger the datasets, the better the model will perform but longer the training will take\n# For the purpose of this tutorial, we'll use a small subset of the dataset\ntraining_size = 3000\nvalidation_size = 300\nDATASET_PATH = Path("dpo-dataset").absolute()\n\n# Get training split and verify it's a Dataset (not IterableDataset)\ntrain_dataset = ds["train"]\nvalidation_dataset = ds["validation"]\nassert isinstance(train_dataset, Dataset), "Expected Dataset type"\nassert isinstance(validation_dataset, Dataset), "Expected Dataset type"\n\n# Select subsets and save to JSONL files\ntesting_ds = train_dataset.select(range(training_size))\nvalidation_ds = validation_dataset.select(range(validation_size))\n\n# Create directory if it doesn't exist\nos.makedirs(DATASET_PATH, exist_ok=True)\n\n# Save subsets to JSONL files\ntesting_ds.to_json(f"{DATASET_PATH}/training.jsonl")\nvalidation_ds.to_json(f"{DATASET_PATH}/validation.jsonl")\n\nprint(f"Saved training.jsonl with {len(testing_ds)} rows")\nprint(f"Saved validation.jsonl with {len(validation_ds)} rows")\n" - }, - { - "type": "markdown", - "source": "#### Upload Training Data", - "source_html": "

Upload Training Data

\n" - }, - { - "type": "code", - "source": "# Create fileset to store DPO training data\nDATASET_NAME = \"dpo-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"dpo training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate training data is uploaded correctly\nprint(\"Training data:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", - "language": "python", - "source_html": "# Create fileset to store DPO training data\nDATASET_NAME = "dpo-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="dpo training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate training data is uploaded correctly\nprint("Training data:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" - }, - { - "type": "markdown", - "source": "### 4. Secrets Setup\n\nIf you plan to use NGC or HuggingFace models, you'll need to configure authentication:\n\n- **NGC models** (`ngc://` URIs): Requires NGC API key\n- **HuggingFace models** (`hf://` URIs): Requires HF token for gated/private models\n\n\nConfigure these as secrets in your platform. See [Managing Secrets](../../get-started/concepts/manage-secrets.md) for detailed instructions.\n\nGet your credentials to access base models:\n- [NGC API Key](https://ngc.nvidia.com/) (Setup → Generate API Key)\n- [HuggingFace Token](https://huggingface.co/settings/tokens) (Create token with Read access)\n\n\n---\n\n#### Quick Setup Example\n\nIn this tutorial we are going to work with [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model from HuggingFace. Ensure that you have sufficient permissions to download the model. If you cannot see the files in the [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) Hugging Face page, request access\n\n**HuggingFace Authentication:**\n- For gated models (Llama, Gemma), you must provide a HuggingFace token via the `token_secret` parameter\n- Get your token from [HuggingFace Settings](https://huggingface.co/settings/tokens) (requires Read access)\n- Accept the model's terms on the HuggingFace model page before using it. Example: [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main)\n- For public models, you can omit the `token_secret` parameter when creating a fileset for model in the next step", - "source_html": "

4. Secrets Setup

\n

If you plan to use NGC or HuggingFace models, you'll need to configure authentication:

\n
    \n
  • NGC models (ngc:// URIs): Requires NGC API key
  • \n
  • HuggingFace models (hf:// URIs): Requires HF token for gated/private models
  • \n
\n

Configure these as secrets in your platform. See Managing Secrets for detailed instructions.

\n

Get your credentials to access base models:

\n\n
\n

Quick Setup Example

\n

In this tutorial we are going to work with meta-llama/Llama-3.2-1B-Instruct model from HuggingFace. Ensure that you have sufficient permissions to download the model. If you cannot see the files in the meta-llama/Llama-3.2-1B-Instruct Hugging Face page, request access

\n

HuggingFace Authentication:

\n
    \n
  • For gated models (Llama, Gemma), you must provide a HuggingFace token via the token_secret parameter
  • \n
  • Get your token from HuggingFace Settings (requires Read access)
  • \n
  • Accept the model's terms on the HuggingFace model page before using it. Example: meta-llama/Llama-3.2-1B-Instruct
  • \n
  • For public models, you can omit the token_secret parameter when creating a fileset for model in the next step
  • \n
\n" - }, - { - "type": "code", - "source": "# Export the HF_TOKEN and NGC_API_KEY environment variables if they are not already set\nHF_TOKEN = os.getenv(\"HF_TOKEN\")\nNGC_API_KEY = os.getenv(\"NGC_API_KEY\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n name=name,\n workspace=\"default\",\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name, workspace=\"default\")\n\n\n# Create HuggingFace token secret\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(\"HF_TOKEN secret:\")\nprint(hf_secret.model_dump_json(indent=2))\n\n# Create NGC API key secret\n# Uncomment the line below if you have NGC API Key and want to finetune NGC models\n# ngc_api_key = create_or_get_secret(\"ngc-api-key\", NGC_API_KEY, \"NGC_API_KEY\")", - "language": "python", - "source_html": "# Export the HF_TOKEN and NGC_API_KEY environment variables if they are not already set\nHF_TOKEN = os.getenv("HF_TOKEN")\nNGC_API_KEY = os.getenv("NGC_API_KEY")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f"{label} is not set")\n try:\n secret = client.secrets.create(\n name=name,\n workspace="default",\n value=value,\n )\n print(f"Created secret: {name}")\n return secret\n except ConflictError:\n print(f"Secret '{name}' already exists, continuing...")\n return client.secrets.retrieve(name=name, workspace="default")\n\n\n# Create HuggingFace token secret\nhf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN")\nprint("HF_TOKEN secret:")\nprint(hf_secret.model_dump_json(indent=2))\n\n# Create NGC API key secret\n# Uncomment the line below if you have NGC API Key and want to finetune NGC models\n# ngc_api_key = create_or_get_secret("ngc-api-key", NGC_API_KEY, "NGC_API_KEY")\n" - }, - { - "type": "markdown", - "source": "### 5. Create Base Model FileSet and Model Entity\n\nCreate a fileset pointing to [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct/tree/main) model in HuggingFace that we will train with DPO. Then create a Model Entity that references this fileset. Model downloading will take place at the DPO finetuning job creation time.\n\nNote: for public models, you can omit the `token_secret` parameter when creating a model fileset.", - "source_html": "

5. Create Base Model FileSet and Model Entity

\n

Create a fileset pointing to meta-llama/Llama-3.2-1B-Instruct model in HuggingFace that we will train with DPO. Then create a Model Entity that references this fileset. Model downloading will take place at the DPO finetuning job creation time.

\n

Note: for public models, you can omit the token_secret parameter when creating a model fileset.

\n" - }, - { - "type": "code", - "source": "import time\n\n# Create a fileset pointing to the desired HuggingFace model\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = \"meta-llama/Llama-3.2-1B-Instruct\"\nMODEL_NAME = \"llama-3-2-1b-base\"\n\n# Ensure you have a HuggingFace token secret created\ntry:\n base_model_fs = client.files.filesets.create(\n workspace=\"default\",\n name=MODEL_NAME,\n description=\"Llama 3.2 1B base model from HuggingFace\",\n storage=HuggingfaceStorageConfigParam(\n type=\"huggingface\",\n # repo_id is the full model name from Hugging Face\n repo_id=HF_REPO_ID,\n repo_type=\"model\",\n # we use the secret created in the previous step\n token_secret=hf_secret.name\n )\n )\n print(f\"Created base model fileset: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model fileset already exists. Skipping creation.\")\n base_model_fs = client.files.filesets.retrieve(\n workspace=\"default\",\n name=MODEL_NAME,\n )\n\n# Create Model Entity referencing the FileSet\ntry:\n base_model = client.models.create(\n workspace=\"default\",\n name=MODEL_NAME,\n fileset=f\"default/{MODEL_NAME}\",\n )\n print(f\"Created Model Entity: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model already exists. Updating fileset if different.\")\n base_model = client.models.update(\n workspace=\"default\",\n name=MODEL_NAME,\n fileset=f\"default/{MODEL_NAME}\",\n )\n\nprint(f\"\\nBase model fileset: fileset://default/{base_model.name}\")\nprint(\"Base model fileset files list:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=\"default\").data], indent=2))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint(\"\\nWaiting for ModelSpec to be populated...\")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n time.sleep(2)\n base_model = client.models.retrieve(\n workspace=\"default\",\n name=MODEL_NAME,\n )\n\nprint(f\"ModelSpec populated: {base_model.spec}\")", - "language": "python", - "source_html": "import time\n\n# Create a fileset pointing to the desired HuggingFace model\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = "meta-llama/Llama-3.2-1B-Instruct"\nMODEL_NAME = "llama-3-2-1b-base"\n\n# Ensure you have a HuggingFace token secret created\ntry:\n base_model_fs = client.files.filesets.create(\n workspace="default",\n name=MODEL_NAME,\n description="Llama 3.2 1B base model from HuggingFace",\n storage=HuggingfaceStorageConfigParam(\n type="huggingface",\n # repo_id is the full model name from Hugging Face\n repo_id=HF_REPO_ID,\n repo_type="model",\n # we use the secret created in the previous step\n token_secret=hf_secret.name\n )\n )\n print(f"Created base model fileset: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model fileset already exists. Skipping creation.")\n base_model_fs = client.files.filesets.retrieve(\n workspace="default",\n name=MODEL_NAME,\n )\n\n# Create Model Entity referencing the FileSet\ntry:\n base_model = client.models.create(\n workspace="default",\n name=MODEL_NAME,\n fileset=f"default/{MODEL_NAME}",\n )\n print(f"Created Model Entity: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model already exists. Updating fileset if different.")\n base_model = client.models.update(\n workspace="default",\n name=MODEL_NAME,\n fileset=f"default/{MODEL_NAME}",\n )\n\nprint(f"\\nBase model fileset: fileset://default/{base_model.name}")\nprint("Base model fileset files list:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace="default").data], indent=2))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint("\\nWaiting for ModelSpec to be populated...")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds")\n time.sleep(2)\n base_model = client.models.retrieve(\n workspace="default",\n name=MODEL_NAME,\n )\n\nprint(f"ModelSpec populated: {base_model.spec}")\n" - }, - { - "type": "markdown", - "source": "### 6. Create DPO Finetuning Job\nCreate a customization job with an inline target referencing the base model and dataset filesets created in previous steps.", - "source_html": "

6. Create DPO Finetuning Job

\n

Create a customization job with an inline target referencing the base model and dataset filesets created in previous steps.

\n" - }, - { - "type": "markdown", - "source": "**Target `model_uri` Format:**\n\nCurrently, `model_uri` must reference a FileSet:\n- **FileSet:** `fileset://{workspace}/{fileset-name}`\n\nSupport for direct HuggingFace (`hf://`) and NGC (`ngc://`) URIs is coming soon. For now, create a fileset as shown in the previous step, and the HuggingFace model will be downloaded at the beginning the finetuning job.\n\n**GPU Requirements:**\n- 1B models: 1 GPU (24GB+ VRAM)\n- 3B models: 1-2 GPUs \n- 8B models: 2-4 GPUs\n- 70B models: 8+ GPUs \n\nAdjust `num_gpus_per_node` and `tensor_parallel_size` based on your model size.\n\n**Important**\n\nWhen setting `val_check_interval` for DPO, use a fractional value (e.g., `0.5` for twice per epoch) or omit it entirely (validates once at end of epoch). Avoid integer step counts — they may not divide evenly into the total training steps, which can prevent validation from running on the final step.", - "source_html": "

Target model_uri Format:

\n

Currently, model_uri must reference a FileSet:

\n
    \n
  • FileSet: fileset://{workspace}/{fileset-name}
  • \n
\n

Support for direct HuggingFace (hf://) and NGC (ngc://) URIs is coming soon. For now, create a fileset as shown in the previous step, and the HuggingFace model will be downloaded at the beginning the finetuning job.

\n

GPU Requirements:

\n
    \n
  • 1B models: 1 GPU (24GB+ VRAM)
  • \n
  • 3B models: 1-2 GPUs
  • \n
  • 8B models: 2-4 GPUs
  • \n
  • 70B models: 8+ GPUs
  • \n
\n

Adjust num_gpus_per_node and tensor_parallel_size based on your model size.

\n

Important

\n

When setting val_check_interval for DPO, use a fractional value (e.g., 0.5 for twice per epoch) or omit it entirely (validates once at end of epoch). Avoid integer step counts — they may not divide evenly into the total training steps, which can prevent validation from running on the final step.

\n" - }, - { - "type": "code", - "source": "import uuid\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f\"my-dpo-job-{job_suffix}\"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=DpoTrainingParam(\n type=\"dpo\",\n epochs=1,\n batch_size=16,\n learning_rate=0.00005,\n max_seq_length=4096,\n ref_policy_kl_penalty=0.1,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n )\n)\n\nprint(f\"Job ID: {job.name}\")\nprint(f\"Output model: {job.spec.output.name}\")", - "language": "python", - "source_html": "import uuid\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f"my-dpo-job-{job_suffix}"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=DpoTrainingParam(\n type="dpo",\n epochs=1,\n batch_size=16,\n learning_rate=0.00005,\n max_seq_length=4096,\n ref_policy_kl_penalty=0.1,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n )\n)\n\nprint(f"Job ID: {job.name}")\nprint(f"Output model: {job.spec.output.name}")\n" - }, - { - "type": "markdown", - "source": "### 7. Track Training Progress", - "source_html": "

7. Track Training Progress

\n" - }, - { - "type": "code", - "source": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace=\"default\"\n )\n \n clear_output(wait=True)\n print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get(\"step\")\n max_steps = task_details.get(\"max_steps\")\n training_phase = task_details.get(\"phase\")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n if training_phase:\n print(f\"Training Phase: {training_phase}\")\n else:\n print(\"Training step not started yet or progress info not available\")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n break\n \n time.sleep(10)", - "language": "python", - "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace="default"\n )\n \n clear_output(wait=True)\n print(f"Job Status: {status.model_dump_json(indent=2)}")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get("step")\n max_steps = task_details.get("max_steps")\n training_phase = task_details.get("phase")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)")\n if training_phase:\n print(f"Training Phase: {training_phase}")\n else:\n print("Training step not started yet or progress info not available")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n break\n \n time.sleep(10)\n" - }, - { - "type": "markdown", - "source": "**Interpreting DPO Training Metrics:**\n\nDPO training produces several key metrics:\n\n| Metric | Description | What to Look For |\n|--------|-------------|------------------|\n| **loss** | Total training loss (preference_loss + sft_loss) | Should decrease over training |\n| **preference_loss** | Core DPO loss measuring preference learning | Starts near ln(2) ≈ 0.693, should decrease |\n| **sft_loss** | SFT regularization term (often 0 for pure DPO) | Depends on configuration |\n| **accuracy** | Fraction of samples where chosen > rejected | Should increase toward 80-95%+ |\n| **rewards_chosen_mean** | Average implicit reward for chosen responses | Should be positive |\n| **rewards_rejected_mean** | Average implicit reward for rejected responses | Should be negative |\n\n**Key Indicators:**\n\n- **Reward Margin** = `rewards_chosen_mean - rewards_rejected_mean`\n - Should be positive and increasing\n - Indicates the model is learning to distinguish preferences\n\n- **Accuracy Interpretation:**\n - 50% = random chance (no learning)\n - 66-75% = early/moderate learning\n - 80%+ = good preference learning\n - 95%+ = strong preference alignment\n\n**Troubleshooting:**\n\n- **Loss near ln(2) ≈ 0.693**: Model is at random chance level, training just starting or not learning\n- **Accuracy stuck at ~50%**: Check data quality, increase learning rate, or verify preference labels\n- **Negative reward margin**: Model is learning the wrong direction—check chosen/rejected labels\n- **Loss increasing**: Learning rate too high or data quality issues\n\n**Note:** Training metrics measure optimization progress, not final model quality. Always evaluate the deployed model on your specific use case.", - "source_html": "

Interpreting DPO Training Metrics:

\n

DPO training produces several key metrics:

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
MetricDescriptionWhat to Look For
lossTotal training loss (preference_loss + sft_loss)Should decrease over training
preference_lossCore DPO loss measuring preference learningStarts near ln(2) ≈ 0.693, should decrease
sft_lossSFT regularization term (often 0 for pure DPO)Depends on configuration
accuracyFraction of samples where chosen > rejectedShould increase toward 80-95%+
rewards_chosen_meanAverage implicit reward for chosen responsesShould be positive
rewards_rejected_meanAverage implicit reward for rejected responsesShould be negative
\n

Key Indicators:

\n
    \n
  • \n

    Reward Margin = rewards_chosen_mean - rewards_rejected_mean

    \n
      \n
    • Should be positive and increasing
    • \n
    • Indicates the model is learning to distinguish preferences
    • \n
    \n
  • \n
  • \n

    Accuracy Interpretation:

    \n
      \n
    • 50% = random chance (no learning)
    • \n
    • 66-75% = early/moderate learning
    • \n
    • 80%+ = good preference learning
    • \n
    • 95%+ = strong preference alignment
    • \n
    \n
  • \n
\n

Troubleshooting:

\n
    \n
  • Loss near ln(2) ≈ 0.693: Model is at random chance level, training just starting or not learning
  • \n
  • Accuracy stuck at ~50%: Check data quality, increase learning rate, or verify preference labels
  • \n
  • Negative reward margin: Model is learning the wrong direction—check chosen/rejected labels
  • \n
  • Loss increasing: Learning rate too high or data quality issues
  • \n
\n

Note: Training metrics measure optimization progress, not final model quality. Always evaluate the deployed model on your specific use case.

\n" - }, - { - "type": "markdown", - "source": "### 8. Deploy Fine-Tuned Model\n\nOnce training completes, deploy using the Deployment Management Service:", - "source_html": "

8. Deploy Fine-Tuned Model

\n

Once training completes, deploy using the Deployment Management Service:

\n" - }, - { - "type": "code", - "source": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))", - "language": "python", - "source_html": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))\n" - }, - { - "type": "code", - "source": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"dpo-model-deployment-cfg-{deploy_suffix}\"\nDEPLOYMENT_NAME = f\"dpo-model-deployment-{deploy_suffix}\"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n image_tag=\"1.13.1\",\n gpu=1,\n model_name=job.spec.output.name, # ModelEntity name from training,\n model_namespace=\"default\", # Workspace where ModelEntity lives\n )\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace=\"default\"\n)\n\nprint(f\"Deployment name: {deployment.name}\")\nprint(f\"Deployment status: {deployment_status.status}\")", - "language": "python", - "source_html": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"dpo-model-deployment-cfg-{deploy_suffix}"\nDEPLOYMENT_NAME = f"dpo-model-deployment-{deploy_suffix}"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name="nvcr.io/nim/nvidia/llm-nim",\n image_tag="1.13.1",\n gpu=1,\n model_name=job.spec.output.name, # ModelEntity name from training,\n model_namespace="default", # Workspace where ModelEntity lives\n )\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace="default"\n)\n\nprint(f"Deployment name: {deployment.name}")\nprint(f"Deployment status: {deployment_status.status}")\n" - }, - { - "type": "markdown", - "source": "### Monitor status of deployment", - "source_html": "

Monitor status of deployment

\n" - }, - { - "type": "code", - "source": "import time\nfrom IPython.display import clear_output\n\n# Poll deployment status every 15 seconds until ready\nTIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f\"Monitoring deployment '{deployment.name}'...\")\nprint(f\"Timeout: {TIMEOUT_MINUTES} minutes\\n\")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace=\"default\"\n )\n \n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n \n clear_output(wait=True)\n print(f\"Deployment: {deployment.name}\")\n print(f\"Status: {deployment_status.status}\")\n print(f\"Elapsed time: {elapsed_min}m {elapsed_sec}s\")\n \n # Check if deployment is ready\n if deployment_status.status == \"READY\":\n print(\"\\nDeployment is ready!\")\n if not client.models.wait_for_gateway(deployment.name, workspace=\"default\", timeout=60):\n raise RuntimeError(\"Inference gateway did not become ready\")\n break\n \n # Check for failure states\n if deployment_status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\"):\n raise RuntimeError(f\"Deployment failed with status: {deployment_status.status}\")\n \n # Check timeout\n if elapsed > timeout_seconds:\n raise TimeoutError(f\"Deployment timeout after {TIMEOUT_MINUTES} minutes\")\n \n time.sleep(15)", - "language": "python", - "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll deployment status every 15 seconds until ready\nTIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f"Monitoring deployment '{deployment.name}'...")\nprint(f"Timeout: {TIMEOUT_MINUTES} minutes\\n")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace="default"\n )\n \n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n \n clear_output(wait=True)\n print(f"Deployment: {deployment.name}")\n print(f"Status: {deployment_status.status}")\n print(f"Elapsed time: {elapsed_min}m {elapsed_sec}s")\n \n # Check if deployment is ready\n if deployment_status.status == "READY":\n print("\\nDeployment is ready!")\n if not client.models.wait_for_gateway(deployment.name, workspace="default", timeout=60):\n raise RuntimeError("Inference gateway did not become ready")\n break\n \n # Check for failure states\n if deployment_status.status in ("FAILED", "ERROR", "TERMINATED", "LOST"):\n raise RuntimeError(f"Deployment failed with status: {deployment_status.status}")\n \n # Check timeout\n if elapsed > timeout_seconds:\n raise TimeoutError(f"Deployment timeout after {TIMEOUT_MINUTES} minutes")\n \n time.sleep(15)\n" - }, - { - "type": "markdown", - "source": "The deployment service automatically:\n- Downloads model weights from the Files service\n- Provisions storage (PVC) for the weights\n- Configures and starts the NIM container\n\n**Multi-GPU Deployment:**\n\nFor larger models requiring multiple GPUs, configure parallelism with environment variables:\n\n```python\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=\"sft-model-config-multigpu\",\n \n nim_deployment={\n \"image_name\": \"nvcr.io/nim/nvidia/llm-nim\",\n \"image_tag\": \"1.13.1\",\n \"gpu\": 2, # Total GPUs\n \"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"2\", # Tensor parallelism\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"1\" # Pipeline parallelism\n }\n }\n)\n```", - "source_html": "

The deployment service automatically:

\n
    \n
  • Downloads model weights from the Files service
  • \n
  • Provisions storage (PVC) for the weights
  • \n
  • Configures and starts the NIM container
  • \n
\n

Multi-GPU Deployment:

\n

For larger models requiring multiple GPUs, configure parallelism with environment variables:

\n
deployment_config = client.inference.deployment_configs.create(\n    workspace="default",\n    name="sft-model-config-multigpu",\n    \n    nim_deployment={\n        "image_name": "nvcr.io/nim/nvidia/llm-nim",\n        "image_tag": "1.13.1",\n        "gpu": 2,  # Total GPUs\n        "additional_envs": {\n            "NIM_TENSOR_PARALLEL_SIZE": "2",  # Tensor parallelism\n            "NIM_PIPELINE_PARALLEL_SIZE": "1"  # Pipeline parallelism\n        }\n    }\n)\n
\n" - }, - { - "type": "markdown", - "source": "**Single-Node Constraint:** Model deployments are limited to a single node. The maximum `gpu` value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.\n\n---\n\n#### GPU Parallelism\n\nBy default, NIM uses all GPUs for tensor parallelism (TP). You can customize this behavior using the `NIM_TENSOR_PARALLEL_SIZE` and `NIM_PIPELINE_PARALLEL_SIZE` environment variables.\n\n| Strategy | Description | Best For |\n|----------|-------------|----------|\n| **Tensor Parallel (TP)** | Splits model layers across GPUs | Lowest latency |\n| **Pipeline Parallel (PP)** | Splits model depth across GPUs | Highest throughput |\n\n**Formula:** `gpu` = `NIM_TENSOR_PARALLEL_SIZE` × `NIM_PIPELINE_PARALLEL_SIZE`\n\n---\n\n#### Example Configurations\n\n**Default (TP=8, PP=1) — Lowest Latency**\n```\n\"gpu\": 8\n# NIM automatically sets NIM_TENSOR_PARALLEL_SIZE=8\n```\n\n**Balanced (TP=4, PP=2)**\n```\n\"gpu\": 8,\n\"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"4\",\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"2\"\n}\n```\n\n**Throughput Optimized (TP=2, PP=4)**\n```\n\"gpu\": 8,\n\"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"2\",\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"4\"\n}\n```", - "source_html": "

Single-Node Constraint: Model deployments are limited to a single node. The maximum gpu value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.

\n
\n

GPU Parallelism

\n

By default, NIM uses all GPUs for tensor parallelism (TP). You can customize this behavior using the NIM_TENSOR_PARALLEL_SIZE and NIM_PIPELINE_PARALLEL_SIZE environment variables.

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
StrategyDescriptionBest For
Tensor Parallel (TP)Splits model layers across GPUsLowest latency
Pipeline Parallel (PP)Splits model depth across GPUsHighest throughput
\n

Formula: gpu = NIM_TENSOR_PARALLEL_SIZE × NIM_PIPELINE_PARALLEL_SIZE

\n
\n

Example Configurations

\n

Default (TP=8, PP=1) — Lowest Latency

\n
"gpu": 8\n# NIM automatically sets NIM_TENSOR_PARALLEL_SIZE=8\n
\n

Balanced (TP=4, PP=2)

\n
"gpu": 8,\n"additional_envs": {\n    "NIM_TENSOR_PARALLEL_SIZE": "4",\n    "NIM_PIPELINE_PARALLEL_SIZE": "2"\n}\n
\n

Throughput Optimized (TP=2, PP=4)

\n
"gpu": 8,\n"additional_envs": {\n    "NIM_TENSOR_PARALLEL_SIZE": "2",\n    "NIM_PIPELINE_PARALLEL_SIZE": "4"\n}\n
\n" - }, - { - "type": "markdown", - "source": "### 9. Evaluate Your Model\n\nAfter training, evaluate whether your model meets your requirements:\n\n#### Quick Manual Evaluation", - "source_html": "

9. Evaluate Your Model

\n

After training, evaluate whether your model meets your requirements:

\n

Quick Manual Evaluation

\n" - }, - { - "type": "code", - "source": "# Wait for deployment to be ready, then test\nmessages = [\n {\"role\": \"system\", \"content\": \"You are a helpful assistant.\"},\n {\"role\": \"user\", \"content\": \"Write a short email to my colleague.\"}\n]\n\nresponse = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": f\"default/{job.spec.output.name}\", # Match the model_name from deployment config\n \"messages\": messages,\n \"temperature\": 0.7,\n \"max_tokens\": 256\n }\n)\n\n# Display prompt and completion\nprint(\"=\" * 60)\nprint(\"PROMPT\")\nprint(\"=\" * 60)\nfor msg in messages:\n print(f\"[{msg['role'].upper()}]\")\n print(msg[\"content\"])\n print()\n\nprint(\"=\" * 60)\nprint(\"COMPLETION\")\nprint(\"=\" * 60)\nprint(\"[ASSISTANT]\")\ncompletion = response[\"choices\"][0][\"message\"][\"content\"]\nprint(completion)", - "language": "python", - "source_html": "# Wait for deployment to be ready, then test\nmessages = [\n {"role": "system", "content": "You are a helpful assistant."},\n {"role": "user", "content": "Write a short email to my colleague."}\n]\n\nresponse = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment.name,\n workspace="default",\n body={\n "model": f"default/{job.spec.output.name}", # Match the model_name from deployment config\n "messages": messages,\n "temperature": 0.7,\n "max_tokens": 256\n }\n)\n\n# Display prompt and completion\nprint("=" * 60)\nprint("PROMPT")\nprint("=" * 60)\nfor msg in messages:\n print(f"[{msg['role'].upper()}]")\n print(msg["content"])\n print()\n\nprint("=" * 60)\nprint("COMPLETION")\nprint("=" * 60)\nprint("[ASSISTANT]")\ncompletion = response["choices"][0]["message"]["content"]\nprint(completion)\n" - }, - { - "type": "markdown", - "source": "#### Evaluation Best Practices\n\n**Manual Evaluation** (Recommended)\n- Test with real-world examples from your use case\n- Compare responses to base model and expected outputs\n- Verify the model exhibits desired behavior changes\n- Check edge cases and error handling\n\n**What to look for:**\n- ✅ Model follows your desired output format\n- ✅ Applies domain knowledge correctly\n- ✅ Maintains general language capabilities\n- ✅ Avoids unwanted behaviors or biases\n- ❌ Doesn't hallucinate facts not in training data\n- ❌ Doesn't produce repetitive or nonsensical outputs\n\n---\n\n## Hyperparameters\n\nFor detailed information on all available hyperparameters, recommended values, and tuning guidance, see the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n\n---\n\n\n## Troubleshooting\n\n**Job fails during model download:**\n- Verify authentication secrets are configured (see [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n- Check the `model_uri` format is correct (`fileset://`)\n- Ensure you have accepted the model's terms of service on HuggingFace\n- Check job status and logs: `client.customization.jobs.retrieve(name=job.name, workspace=\"default\")`\n\n**Job fails with OOM (Out of Memory) error:**\n1. **First try:** Reduce `micro_batch_size` from 2 to 1\n2. **Still OOM:** Reduce `batch_size` from 16 to 8\n3. **Still OOM:** Reduce `max_seq_length` from 2048 to 1024 or 512\n4. **Last resort:** Increase GPU count and use `tensor_parallel_size` for model sharding\n\n**Loss curves not decreasing (underfitting):**\n- Increase training duration: `epochs: 5-10` instead of 3\n- Adjust learning rate: Try `1e-5` to `1e-4`\n- Add warmup: Set `warmup_steps` to ~10% of total training steps\n- Check data quality: Verify formatting, remove duplicates, ensure diversity\n\n**Training loss decreases but validation loss increases (overfitting):**\n- Reduce epochs: Try `epochs: 1-2` instead of 5+\n- Lower learning rate: Use `2e-5` or `1e-5`\n- Increase dataset size and diversity\n- Verify train/validation split has no data leakage\n\n**Model output quality is poor despite good training metrics:**\n- Training metrics optimize for loss, not your actual task—evaluate on real use cases\n- Review data quality, format, and diversity—metrics can be misleading with poor data\n- Try a different base model size or architecture\n- Adjust learning rate and batch size\n- Compare to baseline: Test base model to ensure fine-tuning improved performance\n\n**Deployment fails:**\n- Verify output model exists: `client.models.retrieve(name=job.spec.output.name, workspace=\"default\")`\n- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n- Ensure sufficient GPU resources available for model size\n- Verify NIM image tag `1.13.1` is compatible with your model\n\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning", - "source_html": "

Evaluation Best Practices

\n

Manual Evaluation (Recommended)

\n
    \n
  • Test with real-world examples from your use case
  • \n
  • Compare responses to base model and expected outputs
  • \n
  • Verify the model exhibits desired behavior changes
  • \n
  • Check edge cases and error handling
  • \n
\n

What to look for:

\n
    \n
  • ✅ Model follows your desired output format
  • \n
  • ✅ Applies domain knowledge correctly
  • \n
  • ✅ Maintains general language capabilities
  • \n
  • ✅ Avoids unwanted behaviors or biases
  • \n
  • ❌ Doesn't hallucinate facts not in training data
  • \n
  • ❌ Doesn't produce repetitive or nonsensical outputs
  • \n
\n
\n

Hyperparameters

\n

For detailed information on all available hyperparameters, recommended values, and tuning guidance, see the Hyperparameter Reference.

\n
\n

Troubleshooting

\n

Job fails during model download:

\n
    \n
  • Verify authentication secrets are configured (see Managing Secrets)
  • \n
  • For gated HuggingFace models (Llama, Gemma), accept the license on the model page
  • \n
  • Check the model_uri format is correct (fileset://)
  • \n
  • Ensure you have accepted the model's terms of service on HuggingFace
  • \n
  • Check job status and logs: client.customization.jobs.retrieve(name=job.name, workspace="default")
  • \n
\n

Job fails with OOM (Out of Memory) error:

\n
    \n
  1. First try: Reduce micro_batch_size from 2 to 1
  2. \n
  3. Still OOM: Reduce batch_size from 16 to 8
  4. \n
  5. Still OOM: Reduce max_seq_length from 2048 to 1024 or 512
  6. \n
  7. Last resort: Increase GPU count and use tensor_parallel_size for model sharding
  8. \n
\n

Loss curves not decreasing (underfitting):

\n
    \n
  • Increase training duration: epochs: 5-10 instead of 3
  • \n
  • Adjust learning rate: Try 1e-5 to 1e-4
  • \n
  • Add warmup: Set warmup_steps to ~10% of total training steps
  • \n
  • Check data quality: Verify formatting, remove duplicates, ensure diversity
  • \n
\n

Training loss decreases but validation loss increases (overfitting):

\n
    \n
  • Reduce epochs: Try epochs: 1-2 instead of 5+
  • \n
  • Lower learning rate: Use 2e-5 or 1e-5
  • \n
  • Increase dataset size and diversity
  • \n
  • Verify train/validation split has no data leakage
  • \n
\n

Model output quality is poor despite good training metrics:

\n
    \n
  • Training metrics optimize for loss, not your actual task—evaluate on real use cases
  • \n
  • Review data quality, format, and diversity—metrics can be misleading with poor data
  • \n
  • Try a different base model size or architecture
  • \n
  • Adjust learning rate and batch size
  • \n
  • Compare to baseline: Test base model to ensure fine-tuning improved performance
  • \n
\n

Deployment fails:

\n
    \n
  • Verify output model exists: client.models.retrieve(name=job.spec.output.name, workspace="default")
  • \n
  • Check deployment logs: client.inference.deployments.get_logs(name=deployment.name, workspace="default")
  • \n
  • Ensure sufficient GPU resources available for model size
  • \n
  • Verify NIM image tag 1.13.1 is compatible with your model
  • \n
\n

Next Steps

\n\n" - } -] }; diff --git a/docs/fern/components/notebooks/embedding-customization-job.json b/docs/fern/components/notebooks/embedding-customization-job.json index 21b42099e2..0f4e98cd95 100644 --- a/docs/fern/components/notebooks/embedding-customization-job.json +++ b/docs/fern/components/notebooks/embedding-customization-job.json @@ -7,8 +7,8 @@ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)\n3. **HuggingFace token** with read access to download the SPECTER dataset (get one at [huggingface.co/settings/tokens](https://huggingface.co/settings/tokens))\n4. **NGC API key** to pull NIM container images from nvcr.io (get one at [ngc.nvidia.com](https://ngc.nvidia.com/) → Setup → Generate API Key)", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
  5. HuggingFace token with read access to download the SPECTER dataset (get one at huggingface.co/settings/tokens)
  6. \n
  7. NGC API key to pull NIM container images from nvcr.io (get one at ngc.nvidia.com → Setup → Generate API Key)
  8. \n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n3. **HuggingFace token** with read access to download the SPECTER dataset (get one at [huggingface.co/settings/tokens](https://huggingface.co/settings/tokens))\n4. **NGC API key** to pull NIM container images from nvcr.io (get one at [ngc.nvidia.com](https://ngc.nvidia.com/) → Setup → Generate API Key)", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (PyPI wrapper: pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  4. \n
  5. HuggingFace token with read access to download the SPECTER dataset (get one at huggingface.co/settings/tokens)
  6. \n
  7. NGC API key to pull NIM container images from nvcr.io (get one at ngc.nvidia.com → Setup → Generate API Key)
  8. \n
\n" }, { "type": "markdown", @@ -40,9 +40,9 @@ }, { "type": "code", - "source": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# NGC API key is required to pull NIM images from nvcr.io\nNGC_API_KEY = os.environ.get(\"NGC_API_KEY\")\nif not NGC_API_KEY:\n raise ValueError(\"NGC_API_KEY environment variable is required. Get one at https://ngc.nvidia.com/ → Setup → Generate API Key\")\n\n# Create NGC secret for pulling NIM images\nNGC_SECRET_NAME = \"ngc-api-key\"\ntry:\n client.secrets.create(name=NGC_SECRET_NAME, workspace=\"default\", value=NGC_API_KEY)\n print(f\"Created secret: {NGC_SECRET_NAME}\")\nexcept ConflictError:\n print(f\"Secret '{NGC_SECRET_NAME}' already exists, continuing...\")\n\n# Deploy base model for baseline comparison\nBASE_MODEL_HF = \"nvidia/llama-nemotron-embed-1b-v2\"\nNIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\nNIM_TAG = \"1.13.0\"\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f\"baseline-embedding-cfg-{baseline_suffix}\"\nBASELINE_DEPLOYMENT_NAME = f\"baseline-embedding-{baseline_suffix}\"\n\nprint(\"Creating baseline deployment config...\")\nbaseline_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name=NIM_IMAGE,\n image_tag=NIM_TAG,\n gpu=1,\n image_pull_secret=NGC_SECRET_NAME,\n )\n)\n\nprint(\"Deploying base model...\")\nbaseline_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_config.name\n)\nprint(f\"Baseline deployment: {baseline_deployment.name}\")", + "source": "# NGC API key is required to pull NIM images from nvcr.io\nNGC_API_KEY = os.environ.get(\"NGC_API_KEY\")\nif not NGC_API_KEY:\n raise ValueError(\"NGC_API_KEY environment variable is required. Get one at https://ngc.nvidia.com/ → Setup → Generate API Key\")\n\n# Create NGC secret for pulling NIM images\nNGC_SECRET_NAME = \"ngc-api-key\"\ntry:\n client.secrets.create(name=NGC_SECRET_NAME, workspace=\"default\", value=NGC_API_KEY)\n print(f\"Created secret: {NGC_SECRET_NAME}\")\nexcept ConflictError:\n print(f\"Secret '{NGC_SECRET_NAME}' already exists, continuing...\")\n\n# Deploy base model for baseline comparison\nBASE_MODEL_HF = \"nvidia/llama-nemotron-embed-1b-v2\"\nNIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\nNIM_TAG = \"1.13.0\"\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f\"baseline-embedding-cfg-{baseline_suffix}\"\nBASELINE_DEPLOYMENT_NAME = f\"baseline-embedding-{baseline_suffix}\"\n\nprint(\"Creating baseline deployment config...\")\nbaseline_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_CONFIG,\n engine=\"nim\",\n model_spec={},\n executor_config={\n \"gpu\": 1,\n \"image_name\": NIM_IMAGE,\n \"image_tag\": NIM_TAG,\n },\n)\n\nprint(\"Deploying base model...\")\nbaseline_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_config.name\n)\nprint(f\"Baseline deployment: {baseline_deployment.name}\")", "language": "python", - "source_html": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# NGC API key is required to pull NIM images from nvcr.io\nNGC_API_KEY = os.environ.get("NGC_API_KEY")\nif not NGC_API_KEY:\n raise ValueError("NGC_API_KEY environment variable is required. Get one at https://ngc.nvidia.com/ → Setup → Generate API Key")\n\n# Create NGC secret for pulling NIM images\nNGC_SECRET_NAME = "ngc-api-key"\ntry:\n client.secrets.create(name=NGC_SECRET_NAME, workspace="default", value=NGC_API_KEY)\n print(f"Created secret: {NGC_SECRET_NAME}")\nexcept ConflictError:\n print(f"Secret '{NGC_SECRET_NAME}' already exists, continuing...")\n\n# Deploy base model for baseline comparison\nBASE_MODEL_HF = "nvidia/llama-nemotron-embed-1b-v2"\nNIM_IMAGE = "nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2"\nNIM_TAG = "1.13.0"\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f"baseline-embedding-cfg-{baseline_suffix}"\nBASELINE_DEPLOYMENT_NAME = f"baseline-embedding-{baseline_suffix}"\n\nprint("Creating baseline deployment config...")\nbaseline_config = client.inference.deployment_configs.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name=NIM_IMAGE,\n image_tag=NIM_TAG,\n gpu=1,\n image_pull_secret=NGC_SECRET_NAME,\n )\n)\n\nprint("Deploying base model...")\nbaseline_deployment = client.inference.deployments.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_config.name\n)\nprint(f"Baseline deployment: {baseline_deployment.name}")\n" + "source_html": "# NGC API key is required to pull NIM images from nvcr.io\nNGC_API_KEY = os.environ.get("NGC_API_KEY")\nif not NGC_API_KEY:\n raise ValueError("NGC_API_KEY environment variable is required. Get one at https://ngc.nvidia.com/ → Setup → Generate API Key")\n\n# Create NGC secret for pulling NIM images\nNGC_SECRET_NAME = "ngc-api-key"\ntry:\n client.secrets.create(name=NGC_SECRET_NAME, workspace="default", value=NGC_API_KEY)\n print(f"Created secret: {NGC_SECRET_NAME}")\nexcept ConflictError:\n print(f"Secret '{NGC_SECRET_NAME}' already exists, continuing...")\n\n# Deploy base model for baseline comparison\nBASE_MODEL_HF = "nvidia/llama-nemotron-embed-1b-v2"\nNIM_IMAGE = "nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2"\nNIM_TAG = "1.13.0"\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f"baseline-embedding-cfg-{baseline_suffix}"\nBASELINE_DEPLOYMENT_NAME = f"baseline-embedding-{baseline_suffix}"\n\nprint("Creating baseline deployment config...")\nbaseline_config = client.inference.deployment_configs.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_CONFIG,\n engine="nim",\n model_spec={},\n executor_config={\n "gpu": 1,\n "image_name": NIM_IMAGE,\n "image_tag": NIM_TAG,\n },\n)\n\nprint("Deploying base model...")\nbaseline_deployment = client.inference.deployments.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_config.name\n)\nprint(f"Baseline deployment: {baseline_deployment.name}")\n" }, { "type": "code", @@ -85,9 +85,9 @@ }, { "type": "code", - "source": "# Create fileset to store embedding training data\nDATASET_NAME = \"embedding-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SPECTER embedding training data (scientific paper triplets)\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate upload\nprint(\"\\nUploaded files:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", + "source": "# Create fileset to store embedding training data\nDATASET_NAME = \"embedding-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SPECTER embedding training data (scientific paper triplets)\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files\nclient.files.upload(\n local_path=f\"{DATASET_PATH}/\",\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate upload\nprint(\"\\nUploaded files:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", "language": "python", - "source_html": "# Create fileset to store embedding training data\nDATASET_NAME = "embedding-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SPECTER embedding training data (scientific paper triplets)"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate upload\nprint("\\nUploaded files:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" + "source_html": "# Create fileset to store embedding training data\nDATASET_NAME = "embedding-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SPECTER embedding training data (scientific paper triplets)"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files\nclient.files.upload(\n local_path=f"{DATASET_PATH}/",\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate upload\nprint("\\nUploaded files:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" }, { "type": "markdown", @@ -96,14 +96,14 @@ }, { "type": "code", - "source": "# Create secrets for model access\n# Note: NGC_API_KEY secret was already created in the baseline step (Step 2)\nHF_TOKEN = os.getenv(\"HF_TOKEN\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n name=name,\n workspace=\"default\",\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name, workspace=\"default\")\n\n\n# Create HuggingFace token secret (for downloading model from HF during training)\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(f\"HF_TOKEN secret: {hf_secret.name}\")\n\n# NGC secret was already created in baseline step\nprint(f\"NGC_API_KEY secret: {NGC_SECRET_NAME} (created in Step 2)\")", + "source": "# Create secrets for model access\n# Note: NGC_API_KEY secret was already created in the baseline step (Step 2)\nHF_TOKEN = os.getenv(\"HF_TOKEN\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n name=name,\n workspace=\"default\",\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name, workspace=\"default\")\n\n\n# Create HuggingFace token secret (for downloading model from HF during training)\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(f\"HF_TOKEN secret: {hf_secret.name}\")\n\n# NGC secret was already created in baseline step (Step 2), or use the platform default\nif \"NGC_SECRET_NAME\" not in globals():\n NGC_SECRET_NAME = \"ngc-api-key\"\nprint(f\"NGC_API_KEY secret: {NGC_SECRET_NAME}\")", "language": "python", - "source_html": "# Create secrets for model access\n# Note: NGC_API_KEY secret was already created in the baseline step (Step 2)\nHF_TOKEN = os.getenv("HF_TOKEN")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f"{label} is not set")\n try:\n secret = client.secrets.create(\n name=name,\n workspace="default",\n value=value,\n )\n print(f"Created secret: {name}")\n return secret\n except ConflictError:\n print(f"Secret '{name}' already exists, continuing...")\n return client.secrets.retrieve(name=name, workspace="default")\n\n\n# Create HuggingFace token secret (for downloading model from HF during training)\nhf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN")\nprint(f"HF_TOKEN secret: {hf_secret.name}")\n\n# NGC secret was already created in baseline step\nprint(f"NGC_API_KEY secret: {NGC_SECRET_NAME} (created in Step 2)")\n" + "source_html": "# Create secrets for model access\n# Note: NGC_API_KEY secret was already created in the baseline step (Step 2)\nHF_TOKEN = os.getenv("HF_TOKEN")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f"{label} is not set")\n try:\n secret = client.secrets.create(\n name=name,\n workspace="default",\n value=value,\n )\n print(f"Created secret: {name}")\n return secret\n except ConflictError:\n print(f"Secret '{name}' already exists, continuing...")\n return client.secrets.retrieve(name=name, workspace="default")\n\n\n# Create HuggingFace token secret (for downloading model from HF during training)\nhf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN")\nprint(f"HF_TOKEN secret: {hf_secret.name}")\n\n# NGC secret was already created in baseline step (Step 2), or use the platform default\nif "NGC_SECRET_NAME" not in globals():\n NGC_SECRET_NAME = "ngc-api-key"\nprint(f"NGC_API_KEY secret: {NGC_SECRET_NAME}")\n" }, { "type": "markdown", - "source": "### 7. Create Base Model FileSet and Model Entity\n\nCreate a fileset pointing to the [nvidia/llama-3.2-nv-embedqa-1b-v2](https://huggingface.co/nvidia/llama-3.2-nv-embedqa-1b-v2) embedding model from HuggingFace, then create a Model Entity that references this fileset. Model downloading will take place at training time.\n\n---\n*Note*: Either `MODEL_NAME` or `HF_REPO_ID` below must contain the substring embed to indicate that this is an embedding model.", - "source_html": "

7. Create Base Model FileSet and Model Entity

\n

Create a fileset pointing to the nvidia/llama-3.2-nv-embedqa-1b-v2 embedding model from HuggingFace, then create a Model Entity that references this fileset. Model downloading will take place at training time.

\n
\n

Note: Either MODEL_NAME or HF_REPO_ID below must contain the substring embed to indicate that this is an embedding model.

\n" + "source": "### 7. Create Base Model FileSet and Model Entity\n\nCreate a fileset pointing to the [nvidia/llama-3.2-nv-embedqa-1b-v2](https://huggingface.co/nvidia/llama-3.2-nv-embedqa-1b-v2) embedding model from HuggingFace, then create a Model Entity that references this fileset. Model downloading will take place at training time.", + "source_html": "

7. Create Base Model FileSet and Model Entity

\n

Create a fileset pointing to the nvidia/llama-3.2-nv-embedqa-1b-v2 embedding model from HuggingFace, then create a Model Entity that references this fileset. Model downloading will take place at training time.

\n" }, { "type": "code", @@ -113,14 +113,14 @@ }, { "type": "markdown", - "source": "### 8. Create Embedding Fine-tuning Job\n\nCreate a customization job to fine-tune the embedding model using contrastive learning on the SPECTER dataset.\n\n**Key hyperparameters for embedding fine-tuning:**\n- **`training_type`**: `sft` (supervised fine-tuning)\n- **Full fine-tuning**: No `peft` config needed (omit for all-weights training)\n- **`learning_rate`**: Lower values (1e-6 to 5e-6) work well for embedding models\n- **`batch_size`**: Larger batches improve contrastive learning (128-256 recommended)\n\n**NOTE:**\n\nNeMo Platform does not support unmerged LoRA adapters for embedding models because the embedding NIM requires ONNX format, which cannot represent standalone adapters. This notebook creates a job with all-weights finetuning but you can also run LoRA with `merge=True`, which trains a LoRA adapter and then merges it back into the base model after training. The final output is a standard full-weight checkpoint, identical in format to an all-weights fine-tuned model, but LoRA training is faster, uses less memory, and is more lenient in hyperparameter tuning.\n\nTo do that update the job request like so\n\n```\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=EPOCHS,\n batch_size=BATCH_SIZE,\n learning_rate=LEARNING_RATE,\n max_seq_length=MAX_SEQ_LENGTH,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n peft=LoRaParamsParam(\n type= \"lora\",\n merge=True\n )\n )\n )\n)\n```", - "source_html": "

8. Create Embedding Fine-tuning Job

\n

Create a customization job to fine-tune the embedding model using contrastive learning on the SPECTER dataset.

\n

Key hyperparameters for embedding fine-tuning:

\n
    \n
  • training_type: sft (supervised fine-tuning)
  • \n
  • Full fine-tuning: No peft config needed (omit for all-weights training)
  • \n
  • learning_rate: Lower values (1e-6 to 5e-6) work well for embedding models
  • \n
  • batch_size: Larger batches improve contrastive learning (128-256 recommended)
  • \n
\n

NOTE:

\n

NeMo Platform does not support unmerged LoRA adapters for embedding models because the embedding NIM requires ONNX format, which cannot represent standalone adapters. This notebook creates a job with all-weights finetuning but you can also run LoRA with merge=True, which trains a LoRA adapter and then merges it back into the base model after training. The final output is a standard full-weight checkpoint, identical in format to an all-weights fine-tuned model, but LoRA training is faster, uses less memory, and is more lenient in hyperparameter tuning.

\n

To do that update the job request like so

\n
job = client.customization.jobs.create(\n    name=JOB_NAME,\n    workspace="default",\n    spec=CustomizationJobInputParam(\n        model=f"default/{base_model.name}",\n        dataset=f"fileset://default/{DATASET_NAME}",\n        training=SftTrainingParam(\n            type="sft",\n            epochs=EPOCHS,\n            batch_size=BATCH_SIZE,\n            learning_rate=LEARNING_RATE,\n            max_seq_length=MAX_SEQ_LENGTH,\n            micro_batch_size=1,\n            parallelism=ParallelismParamsParam(\n                num_gpus_per_node=1,\n                num_nodes=1,\n                tensor_parallel_size=1,\n                pipeline_parallel_size=1,\n            ),\n            peft=LoRaParamsParam(\n                type= "lora",\n                merge=True\n            )\n        )\n    )\n)\n
\n" + "source": "### 8. Create Embedding Fine-tuning Job\n\nCreate a customization job to fine-tune the embedding model using contrastive learning on the SPECTER dataset.\n\nSubmit to the **Automodel** backend using `AutomodelJobInput` with split `schedule`, `batch`, `optimizer`, and `parallelism` sections. Reference the model entity and dataset fileset by workspace/name (not `fileset://` URIs).\n\n**Key hyperparameters for embedding fine-tuning:**\n- **`training.training_type`**: `sft`\n- **`training.finetuning_type`**: `all_weights` for full fine-tuning, or `lora_merged` for merged LoRA\n- **`optimizer.learning_rate`**: Lower values (1e-6 to 5e-6) work well for embedding models\n- **`batch.global_batch_size`**: Larger batches improve contrastive learning (128-256 recommended)\n\n**NOTE:**\n\nNeMo Platform does not support unmerged LoRA adapters for embedding models because the embedding NIM requires ONNX format, which cannot represent standalone adapters. This notebook uses all-weights fine-tuning. For merged LoRA, set `finetuning_type` to `lora_merged`:\n\n```python\ntraining={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"lora_merged\",\n \"lora\": {\"rank\": 16, \"alpha\": 32},\n \"max_seq_length\": MAX_SEQ_LENGTH,\n}\n```", + "source_html": "

8. Create Embedding Fine-tuning Job

\n

Create a customization job to fine-tune the embedding model using contrastive learning on the SPECTER dataset.

\n

Submit to the Automodel backend using AutomodelJobInput with split schedule, batch, optimizer, and parallelism sections. Reference the model entity and dataset fileset by workspace/name (not fileset:// URIs).

\n

Key hyperparameters for embedding fine-tuning:

\n
    \n
  • training.training_type: sft
  • \n
  • training.finetuning_type: all_weights for full fine-tuning, or lora_merged for merged LoRA
  • \n
  • optimizer.learning_rate: Lower values (1e-6 to 5e-6) work well for embedding models
  • \n
  • batch.global_batch_size: Larger batches improve contrastive learning (128-256 recommended)
  • \n
\n

NOTE:

\n

NeMo Platform does not support unmerged LoRA adapters for embedding models because the embedding NIM requires ONNX format, which cannot represent standalone adapters. This notebook uses all-weights fine-tuning. For merged LoRA, set finetuning_type to lora_merged:

\n
training={\n    "training_type": "sft",\n    "finetuning_type": "lora_merged",\n    "lora": {"rank": 16, "alpha": 32},\n    "max_seq_length": MAX_SEQ_LENGTH,\n}\n
\n" }, { "type": "code", - "source": "from nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam\n)\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"embedding-finetune-job-{job_suffix}\"\n\n# Hyperparameters optimized for embedding fine-tuning\nEPOCHS = 1\nBATCH_SIZE = 128 # Larger batches help contrastive learning\nLEARNING_RATE = 5e-6 # Lower LR for embedding models\nMAX_SEQ_LENGTH = 512 # Typical for embedding models\n\n# Note: The 'name' field must contain 'embed' for the customizer to detect this as an embedding model\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=EPOCHS,\n batch_size=BATCH_SIZE,\n learning_rate=LEARNING_RATE,\n max_seq_length=MAX_SEQ_LENGTH,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n )\n)\n\nprint(f\"Job ID: {job.name}\")\nprint(f\"Output model: {job.spec.output.name}\")", + "source": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"embedding-finetune-job-{job_suffix}\"\nOUTPUT_NAME = f\"nv-embed-finetuned-{job_suffix}\"\n\nEPOCHS = 1\nBATCH_SIZE = 128\nLEARNING_RATE = 5e-6\nMAX_SEQ_LENGTH = 512\n\nspec = AutomodelJobInput(\n model=f\"default/{base_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"all_weights\",\n \"max_seq_length\": MAX_SEQ_LENGTH,\n },\n schedule={\"epochs\": EPOCHS},\n batch={\"global_batch_size\": BATCH_SIZE, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": LEARNING_RATE},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace=\"default\", name=JOB_NAME\n)\n\nprint(f\"Submitted job: {job.job.name}\")\nprint(f\"Output model: {OUTPUT_NAME}\")", "language": "python", - "source_html": "from nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam\n)\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"embedding-finetune-job-{job_suffix}"\n\n# Hyperparameters optimized for embedding fine-tuning\nEPOCHS = 1\nBATCH_SIZE = 128 # Larger batches help contrastive learning\nLEARNING_RATE = 5e-6 # Lower LR for embedding models\nMAX_SEQ_LENGTH = 512 # Typical for embedding models\n\n# Note: The 'name' field must contain 'embed' for the customizer to detect this as an embedding model\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=EPOCHS,\n batch_size=BATCH_SIZE,\n learning_rate=LEARNING_RATE,\n max_seq_length=MAX_SEQ_LENGTH,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n )\n)\n\nprint(f"Job ID: {job.name}")\nprint(f"Output model: {job.spec.output.name}")\n" + "source_html": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"embedding-finetune-job-{job_suffix}"\nOUTPUT_NAME = f"nv-embed-finetuned-{job_suffix}"\n\nEPOCHS = 1\nBATCH_SIZE = 128\nLEARNING_RATE = 5e-6\nMAX_SEQ_LENGTH = 512\n\nspec = AutomodelJobInput(\n model=f"default/{base_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "all_weights",\n "max_seq_length": MAX_SEQ_LENGTH,\n },\n schedule={"epochs": EPOCHS},\n batch={"global_batch_size": BATCH_SIZE, "micro_batch_size": 1},\n optimizer={"learning_rate": LEARNING_RATE},\n parallelism={"num_gpus_per_node": 1},\n output={"name": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace="default", name=JOB_NAME\n)\n\nprint(f"Submitted job: {job.job.name}")\nprint(f"Output model: {OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -129,9 +129,9 @@ }, { "type": "code", - "source": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace=\"default\"\n )\n \n clear_output(wait=True)\n print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get(\"step\")\n max_steps = task_details.get(\"max_steps\")\n training_phase = task_details.get(\"phase\")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n if training_phase:\n print(f\"Training Phase: {training_phase}\")\n else:\n print(\"Training step not started yet or progress info not available\")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n break\n \n time.sleep(10)", + "source": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.jobs.get_status(\n name=job.job.name,\n workspace=\"default\"\n )\n \n clear_output(wait=True)\n print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == \"training\":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get(\"step\")\n max_steps = task_details.get(\"max_steps\")\n training_phase = task_details.get(\"phase\")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n if training_phase:\n print(f\"Training Phase: {training_phase}\")\n else:\n print(\"Training step not started yet or progress info not available\")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n break\n \n time.sleep(10)", "language": "python", - "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace="default"\n )\n \n clear_output(wait=True)\n print(f"Job Status: {status.model_dump_json(indent=2)}")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get("step")\n max_steps = task_details.get("max_steps")\n training_phase = task_details.get("phase")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)")\n if training_phase:\n print(f"Training Phase: {training_phase}")\n else:\n print("Training step not started yet or progress info not available")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n break\n \n time.sleep(10)\n" + "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.jobs.get_status(\n name=job.job.name,\n workspace="default"\n )\n \n clear_output(wait=True)\n print(f"Job Status: {status.model_dump_json(indent=2)}")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == "training":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get("step")\n max_steps = task_details.get("max_steps")\n training_phase = task_details.get("phase")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)")\n if training_phase:\n print(f"Training Phase: {training_phase}")\n else:\n print("Training step not started yet or progress info not available")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n break\n \n time.sleep(10)\n" }, { "type": "markdown", @@ -145,15 +145,15 @@ }, { "type": "code", - "source": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace=\"default\", name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))", + "source": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace=\"default\", name=OUTPUT_NAME)\nprint(model_entity.model_dump_json(indent=2))", "language": "python", - "source_html": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace="default", name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))\n" + "source_html": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace="default", name=OUTPUT_NAME)\nprint(model_entity.model_dump_json(indent=2))\n" }, { "type": "code", - "source": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config for embedding model\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"embedding-model-deployment-cfg-{deploy_suffix}\"\nDEPLOYMENT_NAME = f\"embedding-model-deployment-{deploy_suffix}\"\n\n# Embedding NIM image\nNIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\nNIM_TAG = \"1.13.0\" # Update if using newer NIM release\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name=NIM_IMAGE,\n image_tag=NIM_TAG,\n gpu=1,\n model_name=job.spec.output.name,\n model_namespace=\"default\",\n )\n)\n\n# Deploy model\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\nprint(f\"Deployment name: {deployment.name}\")\nprint(f\"Deployment status: {client.inference.deployments.retrieve(name=deployment.name, workspace='default').status}\")", + "source": "# Create deployment config for embedding model\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"embedding-model-deployment-cfg-{deploy_suffix}\"\nDEPLOYMENT_NAME = f\"embedding-model-deployment-{deploy_suffix}\"\n\n# Embedding NIM image\nNIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\nNIM_TAG = \"1.13.0\" # Update if using newer NIM release\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n engine=\"nim\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": OUTPUT_NAME,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": NIM_IMAGE,\n \"image_tag\": NIM_TAG,\n },\n)\n\n# Deploy model\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\nprint(f\"Deployment name: {deployment.name}\")\nprint(f\"Deployment status: {client.inference.deployments.retrieve(name=deployment.name, workspace='default').status}\")", "language": "python", - "source_html": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config for embedding model\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"embedding-model-deployment-cfg-{deploy_suffix}"\nDEPLOYMENT_NAME = f"embedding-model-deployment-{deploy_suffix}"\n\n# Embedding NIM image\nNIM_IMAGE = "nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2"\nNIM_TAG = "1.13.0" # Update if using newer NIM release\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name=NIM_IMAGE,\n image_tag=NIM_TAG,\n gpu=1,\n model_name=job.spec.output.name,\n model_namespace="default",\n )\n)\n\n# Deploy model\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\nprint(f"Deployment name: {deployment.name}")\nprint(f"Deployment status: {client.inference.deployments.retrieve(name=deployment.name, workspace='default').status}")\n" + "source_html": "# Create deployment config for embedding model\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"embedding-model-deployment-cfg-{deploy_suffix}"\nDEPLOYMENT_NAME = f"embedding-model-deployment-{deploy_suffix}"\n\n# Embedding NIM image\nNIM_IMAGE = "nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2"\nNIM_TAG = "1.13.0" # Update if using newer NIM release\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n engine="nim",\n model_spec={\n "model_namespace": "default",\n "model_name": OUTPUT_NAME,\n },\n executor_config={\n "gpu": 1,\n "image_name": NIM_IMAGE,\n "image_tag": NIM_TAG,\n },\n)\n\n# Deploy model\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\nprint(f"Deployment name: {deployment.name}")\nprint(f"Deployment status: {client.inference.deployments.retrieve(name=deployment.name, workspace='default').status}")\n" }, { "type": "markdown", @@ -173,9 +173,9 @@ }, { "type": "code", - "source": "# Compare: same query, base model vs fine-tuned\n# Using the same DEMO_QUERY and DEMO_DOCS from the baseline test\nMODEL_ID = f\"default/{job.spec.output.name}\"\n\n# Get query embedding from fine-tuned model\nquery_response = client.inference.gateway.provider.post(\n \"v1/embeddings\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": MODEL_ID,\n \"input\": [DEMO_QUERY],\n \"input_type\": \"query\"\n }\n)\nquery_embedding = query_response[\"data\"][0][\"embedding\"]\n\n# Get document embeddings from fine-tuned model\ndoc_response = client.inference.gateway.provider.post(\n \"v1/embeddings\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": MODEL_ID,\n \"input\": DEMO_DOCS,\n \"input_type\": \"passage\"\n }\n)\ndoc_embeddings = [d[\"embedding\"] for d in doc_response[\"data\"]]\n\n# Calculate similarities and rank\nscores = [(i, cosine_similarity(query_embedding, doc_embeddings[i])) for i in range(len(DEMO_DOCS))]\nFINETUNED_RANKING = sorted(scores, key=lambda x: -x[1])\n\n# Display side-by-side comparison\nprint(f\"Query: \\\"{DEMO_QUERY}\\\"\\n\")\nprint(f\"{'Rank':<6} {'Base Model':<30} {'Fine-tuned Model':<30}\")\nprint(\"-\" * 66)\n\nfor rank in range(len(DEMO_DOCS)):\n b_idx, b_score = BASELINE_RANKING[rank]\n f_idx, f_score = FINETUNED_RANKING[rank]\n \n b_label = f\"{DEMO_LABELS[b_idx]} [{b_score:.3f}]\" + (\" *\" if b_idx in DEMO_RELEVANT else \"\")\n f_label = f\"{DEMO_LABELS[f_idx]} [{f_score:.3f}]\" + (\" *\" if f_idx in DEMO_RELEVANT else \"\")\n \n print(f\"#{rank+1:<5} {b_label:<30} {f_label:<30}\")\n\nprint(\"\\n* = relevant paper\")\nprint(\"\\nThe fine-tuned model pushes 'Random Forest' down and ranks CRF papers higher.\")", + "source": "# Compare: same query, base model vs fine-tuned\n# Using the same DEMO_QUERY and DEMO_DOCS from the baseline test\nMODEL_ID = f\"default/{OUTPUT_NAME}\"\n\n# Get query embedding from fine-tuned model\nquery_response = client.inference.gateway.provider.post(\n \"v1/embeddings\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": MODEL_ID,\n \"input\": [DEMO_QUERY],\n \"input_type\": \"query\"\n }\n)\nquery_embedding = query_response[\"data\"][0][\"embedding\"]\n\n# Get document embeddings from fine-tuned model\ndoc_response = client.inference.gateway.provider.post(\n \"v1/embeddings\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": MODEL_ID,\n \"input\": DEMO_DOCS,\n \"input_type\": \"passage\"\n }\n)\ndoc_embeddings = [d[\"embedding\"] for d in doc_response[\"data\"]]\n\n# Calculate similarities and rank\nscores = [(i, cosine_similarity(query_embedding, doc_embeddings[i])) for i in range(len(DEMO_DOCS))]\nFINETUNED_RANKING = sorted(scores, key=lambda x: -x[1])\n\n# Display side-by-side comparison\nprint(f\"Query: \\\"{DEMO_QUERY}\\\"\\n\")\nprint(f\"{'Rank':<6} {'Base Model':<30} {'Fine-tuned Model':<30}\")\nprint(\"-\" * 66)\n\nfor rank in range(len(DEMO_DOCS)):\n b_idx, b_score = BASELINE_RANKING[rank]\n f_idx, f_score = FINETUNED_RANKING[rank]\n \n b_label = f\"{DEMO_LABELS[b_idx]} [{b_score:.3f}]\" + (\" *\" if b_idx in DEMO_RELEVANT else \"\")\n f_label = f\"{DEMO_LABELS[f_idx]} [{f_score:.3f}]\" + (\" *\" if f_idx in DEMO_RELEVANT else \"\")\n \n print(f\"#{rank+1:<5} {b_label:<30} {f_label:<30}\")\n\nprint(\"\\n* = relevant paper\")\nprint(\"\\nThe fine-tuned model pushes 'Random Forest' down and ranks CRF papers higher.\")", "language": "python", - "source_html": "# Compare: same query, base model vs fine-tuned\n# Using the same DEMO_QUERY and DEMO_DOCS from the baseline test\nMODEL_ID = f"default/{job.spec.output.name}"\n\n# Get query embedding from fine-tuned model\nquery_response = client.inference.gateway.provider.post(\n "v1/embeddings",\n name=deployment.name,\n workspace="default",\n body={\n "model": MODEL_ID,\n "input": [DEMO_QUERY],\n "input_type": "query"\n }\n)\nquery_embedding = query_response["data"][0]["embedding"]\n\n# Get document embeddings from fine-tuned model\ndoc_response = client.inference.gateway.provider.post(\n "v1/embeddings",\n name=deployment.name,\n workspace="default",\n body={\n "model": MODEL_ID,\n "input": DEMO_DOCS,\n "input_type": "passage"\n }\n)\ndoc_embeddings = [d["embedding"] for d in doc_response["data"]]\n\n# Calculate similarities and rank\nscores = [(i, cosine_similarity(query_embedding, doc_embeddings[i])) for i in range(len(DEMO_DOCS))]\nFINETUNED_RANKING = sorted(scores, key=lambda x: -x[1])\n\n# Display side-by-side comparison\nprint(f"Query: \\"{DEMO_QUERY}\\"\\n")\nprint(f"{'Rank':<6} {'Base Model':<30} {'Fine-tuned Model':<30}")\nprint("-" * 66)\n\nfor rank in range(len(DEMO_DOCS)):\n b_idx, b_score = BASELINE_RANKING[rank]\n f_idx, f_score = FINETUNED_RANKING[rank]\n \n b_label = f"{DEMO_LABELS[b_idx]} [{b_score:.3f}]" + (" *" if b_idx in DEMO_RELEVANT else "")\n f_label = f"{DEMO_LABELS[f_idx]} [{f_score:.3f}]" + (" *" if f_idx in DEMO_RELEVANT else "")\n \n print(f"#{rank+1:<5} {b_label:<30} {f_label:<30}")\n\nprint("\\n* = relevant paper")\nprint("\\nThe fine-tuned model pushes 'Random Forest' down and ranks CRF papers higher.")\n" + "source_html": "# Compare: same query, base model vs fine-tuned\n# Using the same DEMO_QUERY and DEMO_DOCS from the baseline test\nMODEL_ID = f"default/{OUTPUT_NAME}"\n\n# Get query embedding from fine-tuned model\nquery_response = client.inference.gateway.provider.post(\n "v1/embeddings",\n name=deployment.name,\n workspace="default",\n body={\n "model": MODEL_ID,\n "input": [DEMO_QUERY],\n "input_type": "query"\n }\n)\nquery_embedding = query_response["data"][0]["embedding"]\n\n# Get document embeddings from fine-tuned model\ndoc_response = client.inference.gateway.provider.post(\n "v1/embeddings",\n name=deployment.name,\n workspace="default",\n body={\n "model": MODEL_ID,\n "input": DEMO_DOCS,\n "input_type": "passage"\n }\n)\ndoc_embeddings = [d["embedding"] for d in doc_response["data"]]\n\n# Calculate similarities and rank\nscores = [(i, cosine_similarity(query_embedding, doc_embeddings[i])) for i in range(len(DEMO_DOCS))]\nFINETUNED_RANKING = sorted(scores, key=lambda x: -x[1])\n\n# Display side-by-side comparison\nprint(f"Query: \\"{DEMO_QUERY}\\"\\n")\nprint(f"{'Rank':<6} {'Base Model':<30} {'Fine-tuned Model':<30}")\nprint("-" * 66)\n\nfor rank in range(len(DEMO_DOCS)):\n b_idx, b_score = BASELINE_RANKING[rank]\n f_idx, f_score = FINETUNED_RANKING[rank]\n \n b_label = f"{DEMO_LABELS[b_idx]} [{b_score:.3f}]" + (" *" if b_idx in DEMO_RELEVANT else "")\n f_label = f"{DEMO_LABELS[f_idx]} [{f_score:.3f}]" + (" *" if f_idx in DEMO_RELEVANT else "")\n \n print(f"#{rank+1:<5} {b_label:<30} {f_label:<30}")\n\nprint("\\n* = relevant paper")\nprint("\\nThe fine-tuned model pushes 'Random Forest' down and ranks CRF papers higher.")\n" }, { "type": "markdown", diff --git a/docs/fern/components/notebooks/embedding-customization-job.ts b/docs/fern/components/notebooks/embedding-customization-job.ts index 36cf919e9f..5b5a3aab41 100644 --- a/docs/fern/components/notebooks/embedding-customization-job.ts +++ b/docs/fern/components/notebooks/embedding-customization-job.ts @@ -1,7 +1,9 @@ -// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -// SPDX-License-Identifier: Apache-2.0 - -/** Auto-generated by ipynb-to-fern-json.py - do not edit */ +/** + * SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + * + * Auto-generated by ipynb-to-fern-json.py - do not edit manually. + */ export default { cells: [ { "type": "markdown", @@ -10,8 +12,8 @@ export default { cells: [ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)\n3. **HuggingFace token** with read access to download the SPECTER dataset (get one at [huggingface.co/settings/tokens](https://huggingface.co/settings/tokens))\n4. **NGC API key** to pull NIM container images from nvcr.io (get one at [ngc.nvidia.com](https://ngc.nvidia.com/) → Setup → Generate API Key)", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
  5. HuggingFace token with read access to download the SPECTER dataset (get one at huggingface.co/settings/tokens)
  6. \n
  7. NGC API key to pull NIM container images from nvcr.io (get one at ngc.nvidia.com → Setup → Generate API Key)
  8. \n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n3. **HuggingFace token** with read access to download the SPECTER dataset (get one at [huggingface.co/settings/tokens](https://huggingface.co/settings/tokens))\n4. **NGC API key** to pull NIM container images from nvcr.io (get one at [ngc.nvidia.com](https://ngc.nvidia.com/) → Setup → Generate API Key)", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (PyPI wrapper: pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  4. \n
  5. HuggingFace token with read access to download the SPECTER dataset (get one at huggingface.co/settings/tokens)
  6. \n
  7. NGC API key to pull NIM container images from nvcr.io (get one at ngc.nvidia.com → Setup → Generate API Key)
  8. \n
\n" }, { "type": "markdown", @@ -43,9 +45,9 @@ export default { cells: [ }, { "type": "code", - "source": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# NGC API key is required to pull NIM images from nvcr.io\nNGC_API_KEY = os.environ.get(\"NGC_API_KEY\")\nif not NGC_API_KEY:\n raise ValueError(\"NGC_API_KEY environment variable is required. Get one at https://ngc.nvidia.com/ → Setup → Generate API Key\")\n\n# Create NGC secret for pulling NIM images\nNGC_SECRET_NAME = \"ngc-api-key\"\ntry:\n client.secrets.create(name=NGC_SECRET_NAME, workspace=\"default\", value=NGC_API_KEY)\n print(f\"Created secret: {NGC_SECRET_NAME}\")\nexcept ConflictError:\n print(f\"Secret '{NGC_SECRET_NAME}' already exists, continuing...\")\n\n# Deploy base model for baseline comparison\nBASE_MODEL_HF = \"nvidia/llama-nemotron-embed-1b-v2\"\nNIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\nNIM_TAG = \"1.13.0\"\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f\"baseline-embedding-cfg-{baseline_suffix}\"\nBASELINE_DEPLOYMENT_NAME = f\"baseline-embedding-{baseline_suffix}\"\n\nprint(\"Creating baseline deployment config...\")\nbaseline_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name=NIM_IMAGE,\n image_tag=NIM_TAG,\n gpu=1,\n image_pull_secret=NGC_SECRET_NAME,\n )\n)\n\nprint(\"Deploying base model...\")\nbaseline_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_config.name\n)\nprint(f\"Baseline deployment: {baseline_deployment.name}\")", + "source": "# NGC API key is required to pull NIM images from nvcr.io\nNGC_API_KEY = os.environ.get(\"NGC_API_KEY\")\nif not NGC_API_KEY:\n raise ValueError(\"NGC_API_KEY environment variable is required. Get one at https://ngc.nvidia.com/ → Setup → Generate API Key\")\n\n# Create NGC secret for pulling NIM images\nNGC_SECRET_NAME = \"ngc-api-key\"\ntry:\n client.secrets.create(name=NGC_SECRET_NAME, workspace=\"default\", value=NGC_API_KEY)\n print(f\"Created secret: {NGC_SECRET_NAME}\")\nexcept ConflictError:\n print(f\"Secret '{NGC_SECRET_NAME}' already exists, continuing...\")\n\n# Deploy base model for baseline comparison\nBASE_MODEL_HF = \"nvidia/llama-nemotron-embed-1b-v2\"\nNIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\nNIM_TAG = \"1.13.0\"\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f\"baseline-embedding-cfg-{baseline_suffix}\"\nBASELINE_DEPLOYMENT_NAME = f\"baseline-embedding-{baseline_suffix}\"\n\nprint(\"Creating baseline deployment config...\")\nbaseline_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_CONFIG,\n engine=\"nim\",\n model_spec={},\n executor_config={\n \"gpu\": 1,\n \"image_name\": NIM_IMAGE,\n \"image_tag\": NIM_TAG,\n },\n)\n\nprint(\"Deploying base model...\")\nbaseline_deployment = client.inference.deployments.create(\n workspace=\"default\",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_config.name\n)\nprint(f\"Baseline deployment: {baseline_deployment.name}\")", "language": "python", - "source_html": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# NGC API key is required to pull NIM images from nvcr.io\nNGC_API_KEY = os.environ.get("NGC_API_KEY")\nif not NGC_API_KEY:\n raise ValueError("NGC_API_KEY environment variable is required. Get one at https://ngc.nvidia.com/ → Setup → Generate API Key")\n\n# Create NGC secret for pulling NIM images\nNGC_SECRET_NAME = "ngc-api-key"\ntry:\n client.secrets.create(name=NGC_SECRET_NAME, workspace="default", value=NGC_API_KEY)\n print(f"Created secret: {NGC_SECRET_NAME}")\nexcept ConflictError:\n print(f"Secret '{NGC_SECRET_NAME}' already exists, continuing...")\n\n# Deploy base model for baseline comparison\nBASE_MODEL_HF = "nvidia/llama-nemotron-embed-1b-v2"\nNIM_IMAGE = "nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2"\nNIM_TAG = "1.13.0"\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f"baseline-embedding-cfg-{baseline_suffix}"\nBASELINE_DEPLOYMENT_NAME = f"baseline-embedding-{baseline_suffix}"\n\nprint("Creating baseline deployment config...")\nbaseline_config = client.inference.deployment_configs.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_CONFIG,\n nim_deployment=NIMDeploymentParam(\n image_name=NIM_IMAGE,\n image_tag=NIM_TAG,\n gpu=1,\n image_pull_secret=NGC_SECRET_NAME,\n )\n)\n\nprint("Deploying base model...")\nbaseline_deployment = client.inference.deployments.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_config.name\n)\nprint(f"Baseline deployment: {baseline_deployment.name}")\n" + "source_html": "# NGC API key is required to pull NIM images from nvcr.io\nNGC_API_KEY = os.environ.get("NGC_API_KEY")\nif not NGC_API_KEY:\n raise ValueError("NGC_API_KEY environment variable is required. Get one at https://ngc.nvidia.com/ → Setup → Generate API Key")\n\n# Create NGC secret for pulling NIM images\nNGC_SECRET_NAME = "ngc-api-key"\ntry:\n client.secrets.create(name=NGC_SECRET_NAME, workspace="default", value=NGC_API_KEY)\n print(f"Created secret: {NGC_SECRET_NAME}")\nexcept ConflictError:\n print(f"Secret '{NGC_SECRET_NAME}' already exists, continuing...")\n\n# Deploy base model for baseline comparison\nBASE_MODEL_HF = "nvidia/llama-nemotron-embed-1b-v2"\nNIM_IMAGE = "nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2"\nNIM_TAG = "1.13.0"\n\nbaseline_suffix = uuid.uuid4().hex[:4]\nBASELINE_DEPLOYMENT_CONFIG = f"baseline-embedding-cfg-{baseline_suffix}"\nBASELINE_DEPLOYMENT_NAME = f"baseline-embedding-{baseline_suffix}"\n\nprint("Creating baseline deployment config...")\nbaseline_config = client.inference.deployment_configs.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_CONFIG,\n engine="nim",\n model_spec={},\n executor_config={\n "gpu": 1,\n "image_name": NIM_IMAGE,\n "image_tag": NIM_TAG,\n },\n)\n\nprint("Deploying base model...")\nbaseline_deployment = client.inference.deployments.create(\n workspace="default",\n name=BASELINE_DEPLOYMENT_NAME,\n config=baseline_config.name\n)\nprint(f"Baseline deployment: {baseline_deployment.name}")\n" }, { "type": "code", @@ -88,9 +90,9 @@ export default { cells: [ }, { "type": "code", - "source": "# Create fileset to store embedding training data\nDATASET_NAME = \"embedding-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SPECTER embedding training data (scientific paper triplets)\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate upload\nprint(\"\\nUploaded files:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", + "source": "# Create fileset to store embedding training data\nDATASET_NAME = \"embedding-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SPECTER embedding training data (scientific paper triplets)\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files\nclient.files.upload(\n local_path=f\"{DATASET_PATH}/\",\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate upload\nprint(\"\\nUploaded files:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", "language": "python", - "source_html": "# Create fileset to store embedding training data\nDATASET_NAME = "embedding-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SPECTER embedding training data (scientific paper triplets)"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate upload\nprint("\\nUploaded files:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" + "source_html": "# Create fileset to store embedding training data\nDATASET_NAME = "embedding-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SPECTER embedding training data (scientific paper triplets)"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files\nclient.files.upload(\n local_path=f"{DATASET_PATH}/",\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate upload\nprint("\\nUploaded files:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" }, { "type": "markdown", @@ -99,14 +101,14 @@ export default { cells: [ }, { "type": "code", - "source": "# Create secrets for model access\n# Note: NGC_API_KEY secret was already created in the baseline step (Step 2)\nHF_TOKEN = os.getenv(\"HF_TOKEN\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n name=name,\n workspace=\"default\",\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name, workspace=\"default\")\n\n\n# Create HuggingFace token secret (for downloading model from HF during training)\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(f\"HF_TOKEN secret: {hf_secret.name}\")\n\n# NGC secret was already created in baseline step\nprint(f\"NGC_API_KEY secret: {NGC_SECRET_NAME} (created in Step 2)\")", + "source": "# Create secrets for model access\n# Note: NGC_API_KEY secret was already created in the baseline step (Step 2)\nHF_TOKEN = os.getenv(\"HF_TOKEN\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n name=name,\n workspace=\"default\",\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name, workspace=\"default\")\n\n\n# Create HuggingFace token secret (for downloading model from HF during training)\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(f\"HF_TOKEN secret: {hf_secret.name}\")\n\n# NGC secret was already created in baseline step (Step 2), or use the platform default\nif \"NGC_SECRET_NAME\" not in globals():\n NGC_SECRET_NAME = \"ngc-api-key\"\nprint(f\"NGC_API_KEY secret: {NGC_SECRET_NAME}\")", "language": "python", - "source_html": "# Create secrets for model access\n# Note: NGC_API_KEY secret was already created in the baseline step (Step 2)\nHF_TOKEN = os.getenv("HF_TOKEN")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f"{label} is not set")\n try:\n secret = client.secrets.create(\n name=name,\n workspace="default",\n value=value,\n )\n print(f"Created secret: {name}")\n return secret\n except ConflictError:\n print(f"Secret '{name}' already exists, continuing...")\n return client.secrets.retrieve(name=name, workspace="default")\n\n\n# Create HuggingFace token secret (for downloading model from HF during training)\nhf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN")\nprint(f"HF_TOKEN secret: {hf_secret.name}")\n\n# NGC secret was already created in baseline step\nprint(f"NGC_API_KEY secret: {NGC_SECRET_NAME} (created in Step 2)")\n" + "source_html": "# Create secrets for model access\n# Note: NGC_API_KEY secret was already created in the baseline step (Step 2)\nHF_TOKEN = os.getenv("HF_TOKEN")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f"{label} is not set")\n try:\n secret = client.secrets.create(\n name=name,\n workspace="default",\n value=value,\n )\n print(f"Created secret: {name}")\n return secret\n except ConflictError:\n print(f"Secret '{name}' already exists, continuing...")\n return client.secrets.retrieve(name=name, workspace="default")\n\n\n# Create HuggingFace token secret (for downloading model from HF during training)\nhf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN")\nprint(f"HF_TOKEN secret: {hf_secret.name}")\n\n# NGC secret was already created in baseline step (Step 2), or use the platform default\nif "NGC_SECRET_NAME" not in globals():\n NGC_SECRET_NAME = "ngc-api-key"\nprint(f"NGC_API_KEY secret: {NGC_SECRET_NAME}")\n" }, { "type": "markdown", - "source": "### 7. Create Base Model FileSet and Model Entity\n\nCreate a fileset pointing to the [nvidia/llama-3.2-nv-embedqa-1b-v2](https://huggingface.co/nvidia/llama-3.2-nv-embedqa-1b-v2) embedding model from HuggingFace, then create a Model Entity that references this fileset. Model downloading will take place at training time.\n\n---\n*Note*: Either `MODEL_NAME` or `HF_REPO_ID` below must contain the substring embed to indicate that this is an embedding model.", - "source_html": "

7. Create Base Model FileSet and Model Entity

\n

Create a fileset pointing to the nvidia/llama-3.2-nv-embedqa-1b-v2 embedding model from HuggingFace, then create a Model Entity that references this fileset. Model downloading will take place at training time.

\n
\n

Note: Either MODEL_NAME or HF_REPO_ID below must contain the substring embed to indicate that this is an embedding model.

\n" + "source": "### 7. Create Base Model FileSet and Model Entity\n\nCreate a fileset pointing to the [nvidia/llama-3.2-nv-embedqa-1b-v2](https://huggingface.co/nvidia/llama-3.2-nv-embedqa-1b-v2) embedding model from HuggingFace, then create a Model Entity that references this fileset. Model downloading will take place at training time.", + "source_html": "

7. Create Base Model FileSet and Model Entity

\n

Create a fileset pointing to the nvidia/llama-3.2-nv-embedqa-1b-v2 embedding model from HuggingFace, then create a Model Entity that references this fileset. Model downloading will take place at training time.

\n" }, { "type": "code", @@ -116,14 +118,14 @@ export default { cells: [ }, { "type": "markdown", - "source": "### 8. Create Embedding Fine-tuning Job\n\nCreate a customization job to fine-tune the embedding model using contrastive learning on the SPECTER dataset.\n\n**Key hyperparameters for embedding fine-tuning:**\n- **`training_type`**: `sft` (supervised fine-tuning)\n- **Full fine-tuning**: No `peft` config needed (omit for all-weights training)\n- **`learning_rate`**: Lower values (1e-6 to 5e-6) work well for embedding models\n- **`batch_size`**: Larger batches improve contrastive learning (128-256 recommended)\n\n**NOTE:**\n\nNeMo Platform does not support unmerged LoRA adapters for embedding models because the embedding NIM requires ONNX format, which cannot represent standalone adapters. This notebook creates a job with all-weights finetuning but you can also run LoRA with `merge=True`, which trains a LoRA adapter and then merges it back into the base model after training. The final output is a standard full-weight checkpoint, identical in format to an all-weights fine-tuned model, but LoRA training is faster, uses less memory, and is more lenient in hyperparameter tuning.\n\nTo do that update the job request like so\n\n```\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=EPOCHS,\n batch_size=BATCH_SIZE,\n learning_rate=LEARNING_RATE,\n max_seq_length=MAX_SEQ_LENGTH,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n peft=LoRaParamsParam(\n type= \"lora\",\n merge=True\n )\n )\n )\n)\n```", - "source_html": "

8. Create Embedding Fine-tuning Job

\n

Create a customization job to fine-tune the embedding model using contrastive learning on the SPECTER dataset.

\n

Key hyperparameters for embedding fine-tuning:

\n
    \n
  • training_type: sft (supervised fine-tuning)
  • \n
  • Full fine-tuning: No peft config needed (omit for all-weights training)
  • \n
  • learning_rate: Lower values (1e-6 to 5e-6) work well for embedding models
  • \n
  • batch_size: Larger batches improve contrastive learning (128-256 recommended)
  • \n
\n

NOTE:

\n

NeMo Platform does not support unmerged LoRA adapters for embedding models because the embedding NIM requires ONNX format, which cannot represent standalone adapters. This notebook creates a job with all-weights finetuning but you can also run LoRA with merge=True, which trains a LoRA adapter and then merges it back into the base model after training. The final output is a standard full-weight checkpoint, identical in format to an all-weights fine-tuned model, but LoRA training is faster, uses less memory, and is more lenient in hyperparameter tuning.

\n

To do that update the job request like so

\n
job = client.customization.jobs.create(\n    name=JOB_NAME,\n    workspace="default",\n    spec=CustomizationJobInputParam(\n        model=f"default/{base_model.name}",\n        dataset=f"fileset://default/{DATASET_NAME}",\n        training=SftTrainingParam(\n            type="sft",\n            epochs=EPOCHS,\n            batch_size=BATCH_SIZE,\n            learning_rate=LEARNING_RATE,\n            max_seq_length=MAX_SEQ_LENGTH,\n            micro_batch_size=1,\n            parallelism=ParallelismParamsParam(\n                num_gpus_per_node=1,\n                num_nodes=1,\n                tensor_parallel_size=1,\n                pipeline_parallel_size=1,\n            ),\n            peft=LoRaParamsParam(\n                type= "lora",\n                merge=True\n            )\n        )\n    )\n)\n
\n" + "source": "### 8. Create Embedding Fine-tuning Job\n\nCreate a customization job to fine-tune the embedding model using contrastive learning on the SPECTER dataset.\n\nSubmit to the **Automodel** backend using `AutomodelJobInput` with split `schedule`, `batch`, `optimizer`, and `parallelism` sections. Reference the model entity and dataset fileset by workspace/name (not `fileset://` URIs).\n\n**Key hyperparameters for embedding fine-tuning:**\n- **`training.training_type`**: `sft`\n- **`training.finetuning_type`**: `all_weights` for full fine-tuning, or `lora_merged` for merged LoRA\n- **`optimizer.learning_rate`**: Lower values (1e-6 to 5e-6) work well for embedding models\n- **`batch.global_batch_size`**: Larger batches improve contrastive learning (128-256 recommended)\n\n**NOTE:**\n\nNeMo Platform does not support unmerged LoRA adapters for embedding models because the embedding NIM requires ONNX format, which cannot represent standalone adapters. This notebook uses all-weights fine-tuning. For merged LoRA, set `finetuning_type` to `lora_merged`:\n\n```python\ntraining={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"lora_merged\",\n \"lora\": {\"rank\": 16, \"alpha\": 32},\n \"max_seq_length\": MAX_SEQ_LENGTH,\n}\n```", + "source_html": "

8. Create Embedding Fine-tuning Job

\n

Create a customization job to fine-tune the embedding model using contrastive learning on the SPECTER dataset.

\n

Submit to the Automodel backend using AutomodelJobInput with split schedule, batch, optimizer, and parallelism sections. Reference the model entity and dataset fileset by workspace/name (not fileset:// URIs).

\n

Key hyperparameters for embedding fine-tuning:

\n
    \n
  • training.training_type: sft
  • \n
  • training.finetuning_type: all_weights for full fine-tuning, or lora_merged for merged LoRA
  • \n
  • optimizer.learning_rate: Lower values (1e-6 to 5e-6) work well for embedding models
  • \n
  • batch.global_batch_size: Larger batches improve contrastive learning (128-256 recommended)
  • \n
\n

NOTE:

\n

NeMo Platform does not support unmerged LoRA adapters for embedding models because the embedding NIM requires ONNX format, which cannot represent standalone adapters. This notebook uses all-weights fine-tuning. For merged LoRA, set finetuning_type to lora_merged:

\n
training={\n    "training_type": "sft",\n    "finetuning_type": "lora_merged",\n    "lora": {"rank": 16, "alpha": 32},\n    "max_seq_length": MAX_SEQ_LENGTH,\n}\n
\n" }, { "type": "code", - "source": "from nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam\n)\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"embedding-finetune-job-{job_suffix}\"\n\n# Hyperparameters optimized for embedding fine-tuning\nEPOCHS = 1\nBATCH_SIZE = 128 # Larger batches help contrastive learning\nLEARNING_RATE = 5e-6 # Lower LR for embedding models\nMAX_SEQ_LENGTH = 512 # Typical for embedding models\n\n# Note: The 'name' field must contain 'embed' for the customizer to detect this as an embedding model\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=EPOCHS,\n batch_size=BATCH_SIZE,\n learning_rate=LEARNING_RATE,\n max_seq_length=MAX_SEQ_LENGTH,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n )\n)\n\nprint(f\"Job ID: {job.name}\")\nprint(f\"Output model: {job.spec.output.name}\")", + "source": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"embedding-finetune-job-{job_suffix}\"\nOUTPUT_NAME = f\"nv-embed-finetuned-{job_suffix}\"\n\nEPOCHS = 1\nBATCH_SIZE = 128\nLEARNING_RATE = 5e-6\nMAX_SEQ_LENGTH = 512\n\nspec = AutomodelJobInput(\n model=f\"default/{base_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"all_weights\",\n \"max_seq_length\": MAX_SEQ_LENGTH,\n },\n schedule={\"epochs\": EPOCHS},\n batch={\"global_batch_size\": BATCH_SIZE, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": LEARNING_RATE},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace=\"default\", name=JOB_NAME\n)\n\nprint(f\"Submitted job: {job.job.name}\")\nprint(f\"Output model: {OUTPUT_NAME}\")", "language": "python", - "source_html": "from nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam\n)\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"embedding-finetune-job-{job_suffix}"\n\n# Hyperparameters optimized for embedding fine-tuning\nEPOCHS = 1\nBATCH_SIZE = 128 # Larger batches help contrastive learning\nLEARNING_RATE = 5e-6 # Lower LR for embedding models\nMAX_SEQ_LENGTH = 512 # Typical for embedding models\n\n# Note: The 'name' field must contain 'embed' for the customizer to detect this as an embedding model\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=EPOCHS,\n batch_size=BATCH_SIZE,\n learning_rate=LEARNING_RATE,\n max_seq_length=MAX_SEQ_LENGTH,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n )\n)\n\nprint(f"Job ID: {job.name}")\nprint(f"Output model: {job.spec.output.name}")\n" + "source_html": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"embedding-finetune-job-{job_suffix}"\nOUTPUT_NAME = f"nv-embed-finetuned-{job_suffix}"\n\nEPOCHS = 1\nBATCH_SIZE = 128\nLEARNING_RATE = 5e-6\nMAX_SEQ_LENGTH = 512\n\nspec = AutomodelJobInput(\n model=f"default/{base_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "all_weights",\n "max_seq_length": MAX_SEQ_LENGTH,\n },\n schedule={"epochs": EPOCHS},\n batch={"global_batch_size": BATCH_SIZE, "micro_batch_size": 1},\n optimizer={"learning_rate": LEARNING_RATE},\n parallelism={"num_gpus_per_node": 1},\n output={"name": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace="default", name=JOB_NAME\n)\n\nprint(f"Submitted job: {job.job.name}")\nprint(f"Output model: {OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -132,9 +134,9 @@ export default { cells: [ }, { "type": "code", - "source": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace=\"default\"\n )\n \n clear_output(wait=True)\n print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get(\"step\")\n max_steps = task_details.get(\"max_steps\")\n training_phase = task_details.get(\"phase\")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n if training_phase:\n print(f\"Training Phase: {training_phase}\")\n else:\n print(\"Training step not started yet or progress info not available\")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n break\n \n time.sleep(10)", + "source": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.jobs.get_status(\n name=job.job.name,\n workspace=\"default\"\n )\n \n clear_output(wait=True)\n print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == \"training\":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get(\"step\")\n max_steps = task_details.get(\"max_steps\")\n training_phase = task_details.get(\"phase\")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n if training_phase:\n print(f\"Training Phase: {training_phase}\")\n else:\n print(\"Training step not started yet or progress info not available\")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n break\n \n time.sleep(10)", "language": "python", - "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace="default"\n )\n \n clear_output(wait=True)\n print(f"Job Status: {status.model_dump_json(indent=2)}")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get("step")\n max_steps = task_details.get("max_steps")\n training_phase = task_details.get("phase")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)")\n if training_phase:\n print(f"Training Phase: {training_phase}")\n else:\n print("Training step not started yet or progress info not available")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n break\n \n time.sleep(10)\n" + "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.jobs.get_status(\n name=job.job.name,\n workspace="default"\n )\n \n clear_output(wait=True)\n print(f"Job Status: {status.model_dump_json(indent=2)}")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == "training":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get("step")\n max_steps = task_details.get("max_steps")\n training_phase = task_details.get("phase")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)")\n if training_phase:\n print(f"Training Phase: {training_phase}")\n else:\n print("Training step not started yet or progress info not available")\n \n # Exit loop when job is completed (or failed/cancelled)\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n break\n \n time.sleep(10)\n" }, { "type": "markdown", @@ -148,15 +150,15 @@ export default { cells: [ }, { "type": "code", - "source": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace=\"default\", name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))", + "source": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace=\"default\", name=OUTPUT_NAME)\nprint(model_entity.model_dump_json(indent=2))", "language": "python", - "source_html": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace="default", name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))\n" + "source_html": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace="default", name=OUTPUT_NAME)\nprint(model_entity.model_dump_json(indent=2))\n" }, { "type": "code", - "source": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config for embedding model\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"embedding-model-deployment-cfg-{deploy_suffix}\"\nDEPLOYMENT_NAME = f\"embedding-model-deployment-{deploy_suffix}\"\n\n# Embedding NIM image\nNIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\nNIM_TAG = \"1.13.0\" # Update if using newer NIM release\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name=NIM_IMAGE,\n image_tag=NIM_TAG,\n gpu=1,\n model_name=job.spec.output.name,\n model_namespace=\"default\",\n )\n)\n\n# Deploy model\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\nprint(f\"Deployment name: {deployment.name}\")\nprint(f\"Deployment status: {client.inference.deployments.retrieve(name=deployment.name, workspace='default').status}\")", + "source": "# Create deployment config for embedding model\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"embedding-model-deployment-cfg-{deploy_suffix}\"\nDEPLOYMENT_NAME = f\"embedding-model-deployment-{deploy_suffix}\"\n\n# Embedding NIM image\nNIM_IMAGE = \"nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2\"\nNIM_TAG = \"1.13.0\" # Update if using newer NIM release\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n engine=\"nim\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": OUTPUT_NAME,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": NIM_IMAGE,\n \"image_tag\": NIM_TAG,\n },\n)\n\n# Deploy model\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\nprint(f\"Deployment name: {deployment.name}\")\nprint(f\"Deployment status: {client.inference.deployments.retrieve(name=deployment.name, workspace='default').status}\")", "language": "python", - "source_html": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config for embedding model\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"embedding-model-deployment-cfg-{deploy_suffix}"\nDEPLOYMENT_NAME = f"embedding-model-deployment-{deploy_suffix}"\n\n# Embedding NIM image\nNIM_IMAGE = "nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2"\nNIM_TAG = "1.13.0" # Update if using newer NIM release\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name=NIM_IMAGE,\n image_tag=NIM_TAG,\n gpu=1,\n model_name=job.spec.output.name,\n model_namespace="default",\n )\n)\n\n# Deploy model\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\nprint(f"Deployment name: {deployment.name}")\nprint(f"Deployment status: {client.inference.deployments.retrieve(name=deployment.name, workspace='default').status}")\n" + "source_html": "# Create deployment config for embedding model\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"embedding-model-deployment-cfg-{deploy_suffix}"\nDEPLOYMENT_NAME = f"embedding-model-deployment-{deploy_suffix}"\n\n# Embedding NIM image\nNIM_IMAGE = "nvcr.io/nim/nvidia/llama-nemotron-embed-1b-v2"\nNIM_TAG = "1.13.0" # Update if using newer NIM release\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n engine="nim",\n model_spec={\n "model_namespace": "default",\n "model_name": OUTPUT_NAME,\n },\n executor_config={\n "gpu": 1,\n "image_name": NIM_IMAGE,\n "image_tag": NIM_TAG,\n },\n)\n\n# Deploy model\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\nprint(f"Deployment name: {deployment.name}")\nprint(f"Deployment status: {client.inference.deployments.retrieve(name=deployment.name, workspace='default').status}")\n" }, { "type": "markdown", @@ -176,9 +178,9 @@ export default { cells: [ }, { "type": "code", - "source": "# Compare: same query, base model vs fine-tuned\n# Using the same DEMO_QUERY and DEMO_DOCS from the baseline test\nMODEL_ID = f\"default/{job.spec.output.name}\"\n\n# Get query embedding from fine-tuned model\nquery_response = client.inference.gateway.provider.post(\n \"v1/embeddings\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": MODEL_ID,\n \"input\": [DEMO_QUERY],\n \"input_type\": \"query\"\n }\n)\nquery_embedding = query_response[\"data\"][0][\"embedding\"]\n\n# Get document embeddings from fine-tuned model\ndoc_response = client.inference.gateway.provider.post(\n \"v1/embeddings\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": MODEL_ID,\n \"input\": DEMO_DOCS,\n \"input_type\": \"passage\"\n }\n)\ndoc_embeddings = [d[\"embedding\"] for d in doc_response[\"data\"]]\n\n# Calculate similarities and rank\nscores = [(i, cosine_similarity(query_embedding, doc_embeddings[i])) for i in range(len(DEMO_DOCS))]\nFINETUNED_RANKING = sorted(scores, key=lambda x: -x[1])\n\n# Display side-by-side comparison\nprint(f\"Query: \\\"{DEMO_QUERY}\\\"\\n\")\nprint(f\"{'Rank':<6} {'Base Model':<30} {'Fine-tuned Model':<30}\")\nprint(\"-\" * 66)\n\nfor rank in range(len(DEMO_DOCS)):\n b_idx, b_score = BASELINE_RANKING[rank]\n f_idx, f_score = FINETUNED_RANKING[rank]\n \n b_label = f\"{DEMO_LABELS[b_idx]} [{b_score:.3f}]\" + (\" *\" if b_idx in DEMO_RELEVANT else \"\")\n f_label = f\"{DEMO_LABELS[f_idx]} [{f_score:.3f}]\" + (\" *\" if f_idx in DEMO_RELEVANT else \"\")\n \n print(f\"#{rank+1:<5} {b_label:<30} {f_label:<30}\")\n\nprint(\"\\n* = relevant paper\")\nprint(\"\\nThe fine-tuned model pushes 'Random Forest' down and ranks CRF papers higher.\")", + "source": "# Compare: same query, base model vs fine-tuned\n# Using the same DEMO_QUERY and DEMO_DOCS from the baseline test\nMODEL_ID = f\"default/{OUTPUT_NAME}\"\n\n# Get query embedding from fine-tuned model\nquery_response = client.inference.gateway.provider.post(\n \"v1/embeddings\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": MODEL_ID,\n \"input\": [DEMO_QUERY],\n \"input_type\": \"query\"\n }\n)\nquery_embedding = query_response[\"data\"][0][\"embedding\"]\n\n# Get document embeddings from fine-tuned model\ndoc_response = client.inference.gateway.provider.post(\n \"v1/embeddings\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": MODEL_ID,\n \"input\": DEMO_DOCS,\n \"input_type\": \"passage\"\n }\n)\ndoc_embeddings = [d[\"embedding\"] for d in doc_response[\"data\"]]\n\n# Calculate similarities and rank\nscores = [(i, cosine_similarity(query_embedding, doc_embeddings[i])) for i in range(len(DEMO_DOCS))]\nFINETUNED_RANKING = sorted(scores, key=lambda x: -x[1])\n\n# Display side-by-side comparison\nprint(f\"Query: \\\"{DEMO_QUERY}\\\"\\n\")\nprint(f\"{'Rank':<6} {'Base Model':<30} {'Fine-tuned Model':<30}\")\nprint(\"-\" * 66)\n\nfor rank in range(len(DEMO_DOCS)):\n b_idx, b_score = BASELINE_RANKING[rank]\n f_idx, f_score = FINETUNED_RANKING[rank]\n \n b_label = f\"{DEMO_LABELS[b_idx]} [{b_score:.3f}]\" + (\" *\" if b_idx in DEMO_RELEVANT else \"\")\n f_label = f\"{DEMO_LABELS[f_idx]} [{f_score:.3f}]\" + (\" *\" if f_idx in DEMO_RELEVANT else \"\")\n \n print(f\"#{rank+1:<5} {b_label:<30} {f_label:<30}\")\n\nprint(\"\\n* = relevant paper\")\nprint(\"\\nThe fine-tuned model pushes 'Random Forest' down and ranks CRF papers higher.\")", "language": "python", - "source_html": "# Compare: same query, base model vs fine-tuned\n# Using the same DEMO_QUERY and DEMO_DOCS from the baseline test\nMODEL_ID = f"default/{job.spec.output.name}"\n\n# Get query embedding from fine-tuned model\nquery_response = client.inference.gateway.provider.post(\n "v1/embeddings",\n name=deployment.name,\n workspace="default",\n body={\n "model": MODEL_ID,\n "input": [DEMO_QUERY],\n "input_type": "query"\n }\n)\nquery_embedding = query_response["data"][0]["embedding"]\n\n# Get document embeddings from fine-tuned model\ndoc_response = client.inference.gateway.provider.post(\n "v1/embeddings",\n name=deployment.name,\n workspace="default",\n body={\n "model": MODEL_ID,\n "input": DEMO_DOCS,\n "input_type": "passage"\n }\n)\ndoc_embeddings = [d["embedding"] for d in doc_response["data"]]\n\n# Calculate similarities and rank\nscores = [(i, cosine_similarity(query_embedding, doc_embeddings[i])) for i in range(len(DEMO_DOCS))]\nFINETUNED_RANKING = sorted(scores, key=lambda x: -x[1])\n\n# Display side-by-side comparison\nprint(f"Query: \\"{DEMO_QUERY}\\"\\n")\nprint(f"{'Rank':<6} {'Base Model':<30} {'Fine-tuned Model':<30}")\nprint("-" * 66)\n\nfor rank in range(len(DEMO_DOCS)):\n b_idx, b_score = BASELINE_RANKING[rank]\n f_idx, f_score = FINETUNED_RANKING[rank]\n \n b_label = f"{DEMO_LABELS[b_idx]} [{b_score:.3f}]" + (" *" if b_idx in DEMO_RELEVANT else "")\n f_label = f"{DEMO_LABELS[f_idx]} [{f_score:.3f}]" + (" *" if f_idx in DEMO_RELEVANT else "")\n \n print(f"#{rank+1:<5} {b_label:<30} {f_label:<30}")\n\nprint("\\n* = relevant paper")\nprint("\\nThe fine-tuned model pushes 'Random Forest' down and ranks CRF papers higher.")\n" + "source_html": "# Compare: same query, base model vs fine-tuned\n# Using the same DEMO_QUERY and DEMO_DOCS from the baseline test\nMODEL_ID = f"default/{OUTPUT_NAME}"\n\n# Get query embedding from fine-tuned model\nquery_response = client.inference.gateway.provider.post(\n "v1/embeddings",\n name=deployment.name,\n workspace="default",\n body={\n "model": MODEL_ID,\n "input": [DEMO_QUERY],\n "input_type": "query"\n }\n)\nquery_embedding = query_response["data"][0]["embedding"]\n\n# Get document embeddings from fine-tuned model\ndoc_response = client.inference.gateway.provider.post(\n "v1/embeddings",\n name=deployment.name,\n workspace="default",\n body={\n "model": MODEL_ID,\n "input": DEMO_DOCS,\n "input_type": "passage"\n }\n)\ndoc_embeddings = [d["embedding"] for d in doc_response["data"]]\n\n# Calculate similarities and rank\nscores = [(i, cosine_similarity(query_embedding, doc_embeddings[i])) for i in range(len(DEMO_DOCS))]\nFINETUNED_RANKING = sorted(scores, key=lambda x: -x[1])\n\n# Display side-by-side comparison\nprint(f"Query: \\"{DEMO_QUERY}\\"\\n")\nprint(f"{'Rank':<6} {'Base Model':<30} {'Fine-tuned Model':<30}")\nprint("-" * 66)\n\nfor rank in range(len(DEMO_DOCS)):\n b_idx, b_score = BASELINE_RANKING[rank]\n f_idx, f_score = FINETUNED_RANKING[rank]\n \n b_label = f"{DEMO_LABELS[b_idx]} [{b_score:.3f}]" + (" *" if b_idx in DEMO_RELEVANT else "")\n f_label = f"{DEMO_LABELS[f_idx]} [{f_score:.3f}]" + (" *" if f_idx in DEMO_RELEVANT else "")\n \n print(f"#{rank+1:<5} {b_label:<30} {f_label:<30}")\n\nprint("\\n* = relevant paper")\nprint("\\nThe fine-tuned model pushes 'Random Forest' down and ranks CRF papers higher.")\n" }, { "type": "markdown", diff --git a/docs/fern/components/notebooks/lora-customization-job.json b/docs/fern/components/notebooks/lora-customization-job.json index bb1ac4e94d..e2f2ff8b16 100644 --- a/docs/fern/components/notebooks/lora-customization-job.json +++ b/docs/fern/components/notebooks/lora-customization-job.json @@ -7,8 +7,8 @@ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)\n3. **Installed the `datasets` package** for loading SQuAD: `pip install datasets`\n4. **At least one GPU with CUDA 12.8+**", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
  5. Installed the datasets package for loading SQuAD: pip install datasets
  6. \n
  7. At least one GPU with CUDA 12.8+
  8. \n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n3. **Installed the `datasets` package** for loading SQuAD: `pip install datasets`\n4. **At least one GPU with CUDA 12.8+**", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (PyPI wrapper: pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  4. \n
  5. Installed the datasets package for loading SQuAD: pip install datasets
  6. \n
  7. At least one GPU with CUDA 12.8+
  8. \n
\n" }, { "type": "markdown", @@ -17,9 +17,9 @@ }, { "type": "code", - "source": "import os\nimport json\nimport re\nimport time\nimport uuid\nfrom pathlib import Path\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.secrets import PlatformSecretResponse\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n DeploymentParamsParam,\n LoRaParamsParam,\n ParallelismParamsParam,\n SftTrainingParam,\n)\n\n\ndef sanitize_name(prefix: str, name: str):\n \"\"\"Sanitize model_name for deployment/config naming. Compatible with platform naming rules.\"\"\"\n name = name.split(\"/\")[-1]\n sanitized = re.sub(r\"[^a-z0-9@.+_-]\", \"-\", name.lower())\n sanitized = re.sub(r\"-+\", \"-\", sanitized).strip(\"-\")\n return f\"{prefix}-{sanitized}\"[:59].rstrip(\"-\")\n\n\ndef max_wait_time_checker(seconds: int, job_name: str = \"\"):\n \"\"\"Return a check() that raises TimeoutError if called after `seconds` have elapsed.\"\"\"\n start_time = time.time()\n\n def check():\n if time.time() - start_time > seconds:\n raise TimeoutError(f\"{job_name} took longer than {seconds} seconds\")\n\n return check\n\n\nNMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace=\"default\"\n)", + "source": "import json\nimport os\nimport re\nimport time\nimport uuid\nfrom pathlib import Path\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.secrets import PlatformSecretResponse\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\n\ndef sanitize_name(prefix: str, name: str):\n \"\"\"Sanitize model_name for deployment/config naming. Compatible with platform naming rules.\"\"\"\n name = name.split(\"/\")[-1]\n sanitized = re.sub(r\"[^a-z0-9@.+_-]\", \"-\", name.lower())\n sanitized = re.sub(r\"-+\", \"-\", sanitized).strip(\"-\")\n return f\"{prefix}-{sanitized}\"[:59].rstrip(\"-\")\n\n\ndef max_wait_time_checker(seconds: int, job_name: str = \"\"):\n \"\"\"Return a check() that raises TimeoutError if called after `seconds` have elapsed.\"\"\"\n start_time = time.time()\n\n def check():\n if time.time() - start_time > seconds:\n raise TimeoutError(f\"{job_name} took longer than {seconds} seconds\")\n\n return check\n\n\nNMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace=\"default\"\n)", "language": "python", - "source_html": "import os\nimport json\nimport re\nimport time\nimport uuid\nfrom pathlib import Path\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.secrets import PlatformSecretResponse\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n DeploymentParamsParam,\n LoRaParamsParam,\n ParallelismParamsParam,\n SftTrainingParam,\n)\n\n\ndef sanitize_name(prefix: str, name: str):\n """Sanitize model_name for deployment/config naming. Compatible with platform naming rules."""\n name = name.split("/")[-1]\n sanitized = re.sub(r"[^a-z0-9@.+_-]", "-", name.lower())\n sanitized = re.sub(r"-+", "-", sanitized).strip("-")\n return f"{prefix}-{sanitized}"[:59].rstrip("-")\n\n\ndef max_wait_time_checker(seconds: int, job_name: str = ""):\n """Return a check() that raises TimeoutError if called after `seconds` have elapsed."""\n start_time = time.time()\n\n def check():\n if time.time() - start_time > seconds:\n raise TimeoutError(f"{job_name} took longer than {seconds} seconds")\n\n return check\n\n\nNMP_BASE_URL = os.environ.get("NMP_BASE_URL", "http://localhost:8080")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace="default"\n)\n" + "source_html": "import json\nimport os\nimport re\nimport time\nimport uuid\nfrom pathlib import Path\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.secrets import PlatformSecretResponse\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\n\ndef sanitize_name(prefix: str, name: str):\n """Sanitize model_name for deployment/config naming. Compatible with platform naming rules."""\n name = name.split("/")[-1]\n sanitized = re.sub(r"[^a-z0-9@.+_-]", "-", name.lower())\n sanitized = re.sub(r"-+", "-", sanitized).strip("-")\n return f"{prefix}-{sanitized}"[:59].rstrip("-")\n\n\ndef max_wait_time_checker(seconds: int, job_name: str = ""):\n """Return a check() that raises TimeoutError if called after `seconds` have elapsed."""\n start_time = time.time()\n\n def check():\n if time.time() - start_time > seconds:\n raise TimeoutError(f"{job_name} took longer than {seconds} seconds")\n\n return check\n\n\nNMP_BASE_URL = os.environ.get("NMP_BASE_URL", "http://localhost:8080")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace="default"\n)\n" }, { "type": "markdown", @@ -77,14 +77,14 @@ }, { "type": "markdown", - "source": "### 6. Create LoRA Customization Job\n\nSubmit a customization job with `training=SftTrainingParam(type=\"sft\", peft=LoRaParamsParam(type=\"lora\"), ...)`. Set `lora_enabled=True` in the `deployment_config` so the platform can deploy the base model with LoRA support automatically.\n\nWhen LoRA Enabled is set to true for Models Deployed via the `/apis/models` endpoint or via the `deployment_config` option during the customization job, all LoRA adapters (enabled by default) will get automatically deployed in the NIM.", - "source_html": "

6. Create LoRA Customization Job

\n

Submit a customization job with training=SftTrainingParam(type="sft", peft=LoRaParamsParam(type="lora"), ...). Set lora_enabled=True in the deployment_config so the platform can deploy the base model with LoRA support automatically.

\n

When LoRA Enabled is set to true for Models Deployed via the /apis/models endpoint or via the deployment_config option during the customization job, all LoRA adapters (enabled by default) will get automatically deployed in the NIM.

\n" + "source": "### 6. Create LoRA Customization Job\n\nSubmit to the **Automodel** backend using `AutomodelJobInput` with `finetuning_type: lora`. After training completes, deploy the base model with LoRA support manually (step 8).", + "source_html": "

6. Create LoRA Customization Job

\n

Submit to the Automodel backend using AutomodelJobInput with finetuning_type: lora. After training completes, deploy the base model with LoRA support manually (step 8).

\n" }, { "type": "code", - "source": "job_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=2,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n context_parallel_size=1,\n expert_parallel_size=1,\n ),\n micro_batch_size=1,\n peft=LoRaParamsParam(type=\"lora\"),\n ),\n deployment_config=DeploymentParamsParam(\n lora_enabled=True,\n gpu=1,\n additional_envs={\"NIM_MODEL_PROFILE\": \"vllm-lora\"},\n ),\n ),\n)\nprint(f\"Job ID: {job.name}\")\nprint(f\"Output model: {job.spec.output.name}\")", + "source": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\nOUTPUT_NAME = f\"lora-adapter-{job_suffix}\"\n\nspec = AutomodelJobInput(\n model=f\"default/{base_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"lora\",\n \"max_seq_length\": 2048,\n },\n schedule={\"epochs\": 2},\n batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\n \"num_gpus_per_node\": 1,\n \"num_nodes\": 1,\n \"tensor_parallel_size\": 1,\n \"pipeline_parallel_size\": 1,\n \"context_parallel_size\": 1,\n \"expert_parallel_size\": 1,\n },\n output={\"name\": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace=\"default\", name=JOB_NAME\n)\nprint(f\"Submitted job: {job.job.name}\")\nprint(f\"Output adapter: {OUTPUT_NAME}\")", "language": "python", - "source_html": "job_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"my-sft-job-{job_suffix}"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=2,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n context_parallel_size=1,\n expert_parallel_size=1,\n ),\n micro_batch_size=1,\n peft=LoRaParamsParam(type="lora"),\n ),\n deployment_config=DeploymentParamsParam(\n lora_enabled=True,\n gpu=1,\n additional_envs={"NIM_MODEL_PROFILE": "vllm-lora"},\n ),\n ),\n)\nprint(f"Job ID: {job.name}")\nprint(f"Output model: {job.spec.output.name}")\n" + "source_html": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"my-sft-job-{job_suffix}"\nOUTPUT_NAME = f"lora-adapter-{job_suffix}"\n\nspec = AutomodelJobInput(\n model=f"default/{base_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "lora",\n "max_seq_length": 2048,\n },\n schedule={"epochs": 2},\n batch={"global_batch_size": 64, "micro_batch_size": 1},\n optimizer={"learning_rate": 5e-5},\n parallelism={\n "num_gpus_per_node": 1,\n "num_nodes": 1,\n "tensor_parallel_size": 1,\n "pipeline_parallel_size": 1,\n "context_parallel_size": 1,\n "expert_parallel_size": 1,\n },\n output={"name": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace="default", name=JOB_NAME\n)\nprint(f"Submitted job: {job.job.name}")\nprint(f"Output adapter: {OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -93,20 +93,20 @@ }, { "type": "code", - "source": "from IPython.display import clear_output\n\ntime_check = max_wait_time_checker(3600, \"Customization Job\")\nwhile True:\n time_check()\n status = client.customization.jobs.get_status(name=job.name, workspace=\"default\")\n clear_output(wait=True)\n print(f\"Job Status: {status.status}\")\n step = max_steps = training_phase = None\n for job_step in status.steps or []:\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n d = task.status_details or {}\n step, max_steps = d.get(\"step\"), d.get(\"max_steps\")\n training_phase = d.get(\"phase\")\n break\n break\n if step is not None and max_steps is not None:\n print(f\"Training: Step {step}/{max_steps} ({100 * step / max_steps:.1f}%)\")\n if training_phase:\n print(f\"Phase: {training_phase}\")\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished: {status.status}\")\n break\n time.sleep(10)\n\nassert status.status == \"completed\"", + "source": "from IPython.display import clear_output\n\ntime_check = max_wait_time_checker(3600, \"Customization Job\")\nwhile True:\n time_check()\n status = client.jobs.get_status(name=job.job.name, workspace=\"default\")\n clear_output(wait=True)\n print(f\"Job Status: {status.status}\")\n step = max_steps = training_phase = None\n for job_step in status.steps or []:\n if job_step.name == \"training\":\n for task in job_step.tasks or []:\n d = task.status_details or {}\n step, max_steps = d.get(\"step\"), d.get(\"max_steps\")\n training_phase = d.get(\"phase\")\n break\n break\n if step is not None and max_steps is not None:\n print(f\"Training: Step {step}/{max_steps} ({100 * step / max_steps:.1f}%)\")\n if training_phase:\n print(f\"Phase: {training_phase}\")\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished: {status.status}\")\n break\n time.sleep(10)\n\nassert status.status == \"completed\"", "language": "python", - "source_html": "from IPython.display import clear_output\n\ntime_check = max_wait_time_checker(3600, "Customization Job")\nwhile True:\n time_check()\n status = client.customization.jobs.get_status(name=job.name, workspace="default")\n clear_output(wait=True)\n print(f"Job Status: {status.status}")\n step = max_steps = training_phase = None\n for job_step in status.steps or []:\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n d = task.status_details or {}\n step, max_steps = d.get("step"), d.get("max_steps")\n training_phase = d.get("phase")\n break\n break\n if step is not None and max_steps is not None:\n print(f"Training: Step {step}/{max_steps} ({100 * step / max_steps:.1f}%)")\n if training_phase:\n print(f"Phase: {training_phase}")\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished: {status.status}")\n break\n time.sleep(10)\n\nassert status.status == "completed"\n" + "source_html": "from IPython.display import clear_output\n\ntime_check = max_wait_time_checker(3600, "Customization Job")\nwhile True:\n time_check()\n status = client.jobs.get_status(name=job.job.name, workspace="default")\n clear_output(wait=True)\n print(f"Job Status: {status.status}")\n step = max_steps = training_phase = None\n for job_step in status.steps or []:\n if job_step.name == "training":\n for task in job_step.tasks or []:\n d = task.status_details or {}\n step, max_steps = d.get("step"), d.get("max_steps")\n training_phase = d.get("phase")\n break\n break\n if step is not None and max_steps is not None:\n print(f"Training: Step {step}/{max_steps} ({100 * step / max_steps:.1f}%)")\n if training_phase:\n print(f"Phase: {training_phase}")\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished: {status.status}")\n break\n time.sleep(10)\n\nassert status.status == "completed"\n" }, { "type": "markdown", - "source": "### 8. Validate Output Model and Deployment\n\nWith `deployment_config` configured, the platform will create a NIM deployment for the base model after training. The fine-tuned LoRA adapter is enabled by default and automatically served through the deployment. Check the model entity and deployment status.", - "source_html": "

8. Validate Output Model and Deployment

\n

With deployment_config configured, the platform will create a NIM deployment for the base model after training. The fine-tuned LoRA adapter is enabled by default and automatically served through the deployment. Check the model entity and deployment status.

\n" + "source": "### 8. Validate Output Model and Deployment\n\nWith the base model entity and LoRA adapter from training, create a NIM deployment with `lora_enabled=True` so the adapter is served alongside the base weights. Check the model entity and deployment status.", + "source_html": "

8. Validate Output Model and Deployment

\n

With the base model entity and LoRA adapter from training, create a NIM deployment with lora_enabled=True so the adapter is served alongside the base weights. Check the model entity and deployment status.

\n" }, { "type": "code", - "source": "model_entity = client.models.retrieve(workspace=\"default\", name=MODEL_NAME)\n# Clear verbose linear_layers list for cleaner output\nmodel_entity.spec.linear_layers = None\nprint(model_entity.model_dump_json(indent=2))\n\ndeployment_name = sanitize_name(\"sft-deploy\", job.spec.model)\ndeployment_status = client.inference.deployments.retrieve(name=deployment_name, workspace=\"default\")\nprint(f\"Deployment status: {deployment_status.status}\")", + "source": "deploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"lora-deploy-cfg-{deploy_suffix}\"\ndeployment_name = f\"lora-deploy-{deploy_suffix}\"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": MODEL_NAME,\n \"lora_enabled\": True,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n \"additional_args\": [\"--max-lora-rank\", \"32\"],\n },\n)\n\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=deployment_name,\n config=deployment_config.name,\n)\n\nmodel_entity = client.models.retrieve(workspace=\"default\", name=MODEL_NAME)\nif model_entity.spec:\n model_entity.spec.linear_layers = None\nprint(model_entity.model_dump_json(indent=2))\nprint(f\"Deployment status: {deployment.status}\")", "language": "python", - "source_html": "model_entity = client.models.retrieve(workspace="default", name=MODEL_NAME)\n# Clear verbose linear_layers list for cleaner output\nmodel_entity.spec.linear_layers = None\nprint(model_entity.model_dump_json(indent=2))\n\ndeployment_name = sanitize_name("sft-deploy", job.spec.model)\ndeployment_status = client.inference.deployments.retrieve(name=deployment_name, workspace="default")\nprint(f"Deployment status: {deployment_status.status}")\n" + "source_html": "deploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"lora-deploy-cfg-{deploy_suffix}"\ndeployment_name = f"lora-deploy-{deploy_suffix}"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n engine="vllm",\n model_spec={\n "model_namespace": "default",\n "model_name": MODEL_NAME,\n "lora_enabled": True,\n },\n executor_config={\n "gpu": 1,\n "image_name": "vllm/vllm-openai",\n "image_tag": "v0.22.1",\n "additional_args": ["--max-lora-rank", "32"],\n },\n)\n\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=deployment_name,\n config=deployment_config.name,\n)\n\nmodel_entity = client.models.retrieve(workspace="default", name=MODEL_NAME)\nif model_entity.spec:\n model_entity.spec.linear_layers = None\nprint(model_entity.model_dump_json(indent=2))\nprint(f"Deployment status: {deployment.status}")\n" }, { "type": "markdown", @@ -126,14 +126,14 @@ }, { "type": "code", - "source": "context = \"The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit.\"\nquestion = \"Who was the first person to walk on the Moon?\"\nmessages = [\n {\"role\": \"user\", \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\"}\n]\nresponse = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment_name,\n workspace=\"default\",\n body={\n \"model\": job.spec.output.name,\n \"messages\": messages,\n \"temperature\": 0,\n \"max_tokens\": 256,\n }\n)\nprint(\"=\" * 60)\nprint(\"MODEL INFERENCE\")\nprint(\"=\" * 60)\nprint(f\"Question: {question}\")\nprint(f\"Expected: Neil Armstrong\")\nprint(f\"Model output: {response['choices'][0]['message']['content']}\")", + "source": "context = \"The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit.\"\nquestion = \"Who was the first person to walk on the Moon?\"\nmessages = [\n {\"role\": \"user\", \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\"}\n]\nresponse = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment_name,\n workspace=\"default\",\n body={\n \"model\": OUTPUT_NAME,\n \"messages\": messages,\n \"temperature\": 0,\n \"max_tokens\": 256,\n }\n)\nprint(\"=\" * 60)\nprint(\"MODEL INFERENCE\")\nprint(\"=\" * 60)\nprint(f\"Question: {question}\")\nprint(f\"Expected: Neil Armstrong\")\nprint(f\"Model output: {response['choices'][0]['message']['content']}\")", "language": "python", - "source_html": "context = "The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit."\nquestion = "Who was the first person to walk on the Moon?"\nmessages = [\n {"role": "user", "content": f"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}"}\n]\nresponse = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment_name,\n workspace="default",\n body={\n "model": job.spec.output.name,\n "messages": messages,\n "temperature": 0,\n "max_tokens": 256,\n }\n)\nprint("=" * 60)\nprint("MODEL INFERENCE")\nprint("=" * 60)\nprint(f"Question: {question}")\nprint(f"Expected: Neil Armstrong")\nprint(f"Model output: {response['choices'][0]['message']['content']}")\n" + "source_html": "context = "The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit."\nquestion = "Who was the first person to walk on the Moon?"\nmessages = [\n {"role": "user", "content": f"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}"}\n]\nresponse = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment_name,\n workspace="default",\n body={\n "model": OUTPUT_NAME,\n "messages": messages,\n "temperature": 0,\n "max_tokens": 256,\n }\n)\nprint("=" * 60)\nprint("MODEL INFERENCE")\nprint("=" * 60)\nprint(f"Question: {question}")\nprint(f"Expected: Neil Armstrong")\nprint(f"Model output: {response['choices'][0]['message']['content']}")\n" }, { "type": "markdown", - "source": "## Conclusion\n\nYou have started a LoRA customization job, monitored it to completion, and evaluated the fine-tuned model. Use the `output.name` to access the model for further inference or evaluation.\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Try [Full SFT](./sft-customization-job) or [DPO](./dpo-customization-job) for other customization options", - "source_html": "

Conclusion

\n

You have started a LoRA customization job, monitored it to completion, and evaluated the fine-tuned model. Use the output.name to access the model for further inference or evaluation.

\n

Next Steps

\n\n" + "source": "## Conclusion\n\nYou have started a LoRA customization job, monitored it to completion, and evaluated the fine-tuned model. Use the `output.name` to access the model for further inference or evaluation.\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Try [Full SFT](./sft-customization-job) for other customization options", + "source_html": "

Conclusion

\n

You have started a LoRA customization job, monitored it to completion, and evaluated the fine-tuned model. Use the output.name to access the model for further inference or evaluation.

\n

Next Steps

\n\n" } ] } \ No newline at end of file diff --git a/docs/fern/components/notebooks/lora-customization-job.ts b/docs/fern/components/notebooks/lora-customization-job.ts index 3f1aacbf4e..d4b0272cd4 100644 --- a/docs/fern/components/notebooks/lora-customization-job.ts +++ b/docs/fern/components/notebooks/lora-customization-job.ts @@ -1,7 +1,9 @@ -// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -// SPDX-License-Identifier: Apache-2.0 - -/** Auto-generated by ipynb-to-fern-json.py - do not edit */ +/** + * SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + * + * Auto-generated by ipynb-to-fern-json.py - do not edit manually. + */ export default { cells: [ { "type": "markdown", @@ -10,8 +12,8 @@ export default { cells: [ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)\n3. **Installed the `datasets` package** for loading SQuAD: `pip install datasets`\n4. **At least one GPU with CUDA 12.8+**", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
  5. Installed the datasets package for loading SQuAD: pip install datasets
  6. \n
  7. At least one GPU with CUDA 12.8+
  8. \n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n3. **Installed the `datasets` package** for loading SQuAD: `pip install datasets`\n4. **At least one GPU with CUDA 12.8+**", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (PyPI wrapper: pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  4. \n
  5. Installed the datasets package for loading SQuAD: pip install datasets
  6. \n
  7. At least one GPU with CUDA 12.8+
  8. \n
\n" }, { "type": "markdown", @@ -20,9 +22,9 @@ export default { cells: [ }, { "type": "code", - "source": "import os\nimport json\nimport re\nimport time\nimport uuid\nfrom pathlib import Path\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.secrets import PlatformSecretResponse\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n DeploymentParamsParam,\n LoRaParamsParam,\n ParallelismParamsParam,\n SftTrainingParam,\n)\n\n\ndef sanitize_name(prefix: str, name: str):\n \"\"\"Sanitize model_name for deployment/config naming. Compatible with platform naming rules.\"\"\"\n name = name.split(\"/\")[-1]\n sanitized = re.sub(r\"[^a-z0-9@.+_-]\", \"-\", name.lower())\n sanitized = re.sub(r\"-+\", \"-\", sanitized).strip(\"-\")\n return f\"{prefix}-{sanitized}\"[:59].rstrip(\"-\")\n\n\ndef max_wait_time_checker(seconds: int, job_name: str = \"\"):\n \"\"\"Return a check() that raises TimeoutError if called after `seconds` have elapsed.\"\"\"\n start_time = time.time()\n\n def check():\n if time.time() - start_time > seconds:\n raise TimeoutError(f\"{job_name} took longer than {seconds} seconds\")\n\n return check\n\n\nNMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace=\"default\"\n)", + "source": "import json\nimport os\nimport re\nimport time\nimport uuid\nfrom pathlib import Path\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.secrets import PlatformSecretResponse\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\n\ndef sanitize_name(prefix: str, name: str):\n \"\"\"Sanitize model_name for deployment/config naming. Compatible with platform naming rules.\"\"\"\n name = name.split(\"/\")[-1]\n sanitized = re.sub(r\"[^a-z0-9@.+_-]\", \"-\", name.lower())\n sanitized = re.sub(r\"-+\", \"-\", sanitized).strip(\"-\")\n return f\"{prefix}-{sanitized}\"[:59].rstrip(\"-\")\n\n\ndef max_wait_time_checker(seconds: int, job_name: str = \"\"):\n \"\"\"Return a check() that raises TimeoutError if called after `seconds` have elapsed.\"\"\"\n start_time = time.time()\n\n def check():\n if time.time() - start_time > seconds:\n raise TimeoutError(f\"{job_name} took longer than {seconds} seconds\")\n\n return check\n\n\nNMP_BASE_URL = os.environ.get(\"NMP_BASE_URL\", \"http://localhost:8080\")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace=\"default\"\n)", "language": "python", - "source_html": "import os\nimport json\nimport re\nimport time\nimport uuid\nfrom pathlib import Path\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.secrets import PlatformSecretResponse\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n DeploymentParamsParam,\n LoRaParamsParam,\n ParallelismParamsParam,\n SftTrainingParam,\n)\n\n\ndef sanitize_name(prefix: str, name: str):\n """Sanitize model_name for deployment/config naming. Compatible with platform naming rules."""\n name = name.split("/")[-1]\n sanitized = re.sub(r"[^a-z0-9@.+_-]", "-", name.lower())\n sanitized = re.sub(r"-+", "-", sanitized).strip("-")\n return f"{prefix}-{sanitized}"[:59].rstrip("-")\n\n\ndef max_wait_time_checker(seconds: int, job_name: str = ""):\n """Return a check() that raises TimeoutError if called after `seconds` have elapsed."""\n start_time = time.time()\n\n def check():\n if time.time() - start_time > seconds:\n raise TimeoutError(f"{job_name} took longer than {seconds} seconds")\n\n return check\n\n\nNMP_BASE_URL = os.environ.get("NMP_BASE_URL", "http://localhost:8080")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace="default"\n)\n" + "source_html": "import json\nimport os\nimport re\nimport time\nimport uuid\nfrom pathlib import Path\nfrom nemo_platform import NeMoPlatform, ConflictError\nfrom nemo_platform.types.secrets import PlatformSecretResponse\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\n\ndef sanitize_name(prefix: str, name: str):\n """Sanitize model_name for deployment/config naming. Compatible with platform naming rules."""\n name = name.split("/")[-1]\n sanitized = re.sub(r"[^a-z0-9@.+_-]", "-", name.lower())\n sanitized = re.sub(r"-+", "-", sanitized).strip("-")\n return f"{prefix}-{sanitized}"[:59].rstrip("-")\n\n\ndef max_wait_time_checker(seconds: int, job_name: str = ""):\n """Return a check() that raises TimeoutError if called after `seconds` have elapsed."""\n start_time = time.time()\n\n def check():\n if time.time() - start_time > seconds:\n raise TimeoutError(f"{job_name} took longer than {seconds} seconds")\n\n return check\n\n\nNMP_BASE_URL = os.environ.get("NMP_BASE_URL", "http://localhost:8080")\nclient = NeMoPlatform(\n base_url=NMP_BASE_URL,\n workspace="default"\n)\n" }, { "type": "markdown", @@ -80,14 +82,14 @@ export default { cells: [ }, { "type": "markdown", - "source": "### 6. Create LoRA Customization Job\n\nSubmit a customization job with `training=SftTrainingParam(type=\"sft\", peft=LoRaParamsParam(type=\"lora\"), ...)`. Set `lora_enabled=True` in the `deployment_config` so the platform can deploy the base model with LoRA support automatically.\n\nWhen LoRA Enabled is set to true for Models Deployed via the `/apis/models` endpoint or via the `deployment_config` option during the customization job, all LoRA adapters (enabled by default) will get automatically deployed in the NIM.", - "source_html": "

6. Create LoRA Customization Job

\n

Submit a customization job with training=SftTrainingParam(type="sft", peft=LoRaParamsParam(type="lora"), ...). Set lora_enabled=True in the deployment_config so the platform can deploy the base model with LoRA support automatically.

\n

When LoRA Enabled is set to true for Models Deployed via the /apis/models endpoint or via the deployment_config option during the customization job, all LoRA adapters (enabled by default) will get automatically deployed in the NIM.

\n" + "source": "### 6. Create LoRA Customization Job\n\nSubmit to the **Automodel** backend using `AutomodelJobInput` with `finetuning_type: lora`. After training completes, deploy the base model with LoRA support manually (step 8).", + "source_html": "

6. Create LoRA Customization Job

\n

Submit to the Automodel backend using AutomodelJobInput with finetuning_type: lora. After training completes, deploy the base model with LoRA support manually (step 8).

\n" }, { "type": "code", - "source": "job_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=2,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n context_parallel_size=1,\n expert_parallel_size=1,\n ),\n micro_batch_size=1,\n peft=LoRaParamsParam(type=\"lora\"),\n ),\n deployment_config=DeploymentParamsParam(\n lora_enabled=True,\n gpu=1,\n additional_envs={\"NIM_MODEL_PROFILE\": \"vllm-lora\"},\n ),\n ),\n)\nprint(f\"Job ID: {job.name}\")\nprint(f\"Output model: {job.spec.output.name}\")", + "source": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\nOUTPUT_NAME = f\"lora-adapter-{job_suffix}\"\n\nspec = AutomodelJobInput(\n model=f\"default/{base_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"lora\",\n \"max_seq_length\": 2048,\n },\n schedule={\"epochs\": 2},\n batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\n \"num_gpus_per_node\": 1,\n \"num_nodes\": 1,\n \"tensor_parallel_size\": 1,\n \"pipeline_parallel_size\": 1,\n \"context_parallel_size\": 1,\n \"expert_parallel_size\": 1,\n },\n output={\"name\": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace=\"default\", name=JOB_NAME\n)\nprint(f\"Submitted job: {job.job.name}\")\nprint(f\"Output adapter: {OUTPUT_NAME}\")", "language": "python", - "source_html": "job_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"my-sft-job-{job_suffix}"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=2,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n context_parallel_size=1,\n expert_parallel_size=1,\n ),\n micro_batch_size=1,\n peft=LoRaParamsParam(type="lora"),\n ),\n deployment_config=DeploymentParamsParam(\n lora_enabled=True,\n gpu=1,\n additional_envs={"NIM_MODEL_PROFILE": "vllm-lora"},\n ),\n ),\n)\nprint(f"Job ID: {job.name}")\nprint(f"Output model: {job.spec.output.name}")\n" + "source_html": "from nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"my-sft-job-{job_suffix}"\nOUTPUT_NAME = f"lora-adapter-{job_suffix}"\n\nspec = AutomodelJobInput(\n model=f"default/{base_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "lora",\n "max_seq_length": 2048,\n },\n schedule={"epochs": 2},\n batch={"global_batch_size": 64, "micro_batch_size": 1},\n optimizer={"learning_rate": 5e-5},\n parallelism={\n "num_gpus_per_node": 1,\n "num_nodes": 1,\n "tensor_parallel_size": 1,\n "pipeline_parallel_size": 1,\n "context_parallel_size": 1,\n "expert_parallel_size": 1,\n },\n output={"name": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace="default", name=JOB_NAME\n)\nprint(f"Submitted job: {job.job.name}")\nprint(f"Output adapter: {OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -96,20 +98,20 @@ export default { cells: [ }, { "type": "code", - "source": "from IPython.display import clear_output\n\ntime_check = max_wait_time_checker(3600, \"Customization Job\")\nwhile True:\n time_check()\n status = client.customization.jobs.get_status(name=job.name, workspace=\"default\")\n clear_output(wait=True)\n print(f\"Job Status: {status.status}\")\n step = max_steps = training_phase = None\n for job_step in status.steps or []:\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n d = task.status_details or {}\n step, max_steps = d.get(\"step\"), d.get(\"max_steps\")\n training_phase = d.get(\"phase\")\n break\n break\n if step is not None and max_steps is not None:\n print(f\"Training: Step {step}/{max_steps} ({100 * step / max_steps:.1f}%)\")\n if training_phase:\n print(f\"Phase: {training_phase}\")\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished: {status.status}\")\n break\n time.sleep(10)\n\nassert status.status == \"completed\"", + "source": "from IPython.display import clear_output\n\ntime_check = max_wait_time_checker(3600, \"Customization Job\")\nwhile True:\n time_check()\n status = client.jobs.get_status(name=job.job.name, workspace=\"default\")\n clear_output(wait=True)\n print(f\"Job Status: {status.status}\")\n step = max_steps = training_phase = None\n for job_step in status.steps or []:\n if job_step.name == \"training\":\n for task in job_step.tasks or []:\n d = task.status_details or {}\n step, max_steps = d.get(\"step\"), d.get(\"max_steps\")\n training_phase = d.get(\"phase\")\n break\n break\n if step is not None and max_steps is not None:\n print(f\"Training: Step {step}/{max_steps} ({100 * step / max_steps:.1f}%)\")\n if training_phase:\n print(f\"Phase: {training_phase}\")\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished: {status.status}\")\n break\n time.sleep(10)\n\nassert status.status == \"completed\"", "language": "python", - "source_html": "from IPython.display import clear_output\n\ntime_check = max_wait_time_checker(3600, "Customization Job")\nwhile True:\n time_check()\n status = client.customization.jobs.get_status(name=job.name, workspace="default")\n clear_output(wait=True)\n print(f"Job Status: {status.status}")\n step = max_steps = training_phase = None\n for job_step in status.steps or []:\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n d = task.status_details or {}\n step, max_steps = d.get("step"), d.get("max_steps")\n training_phase = d.get("phase")\n break\n break\n if step is not None and max_steps is not None:\n print(f"Training: Step {step}/{max_steps} ({100 * step / max_steps:.1f}%)")\n if training_phase:\n print(f"Phase: {training_phase}")\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished: {status.status}")\n break\n time.sleep(10)\n\nassert status.status == "completed"\n" + "source_html": "from IPython.display import clear_output\n\ntime_check = max_wait_time_checker(3600, "Customization Job")\nwhile True:\n time_check()\n status = client.jobs.get_status(name=job.job.name, workspace="default")\n clear_output(wait=True)\n print(f"Job Status: {status.status}")\n step = max_steps = training_phase = None\n for job_step in status.steps or []:\n if job_step.name == "training":\n for task in job_step.tasks or []:\n d = task.status_details or {}\n step, max_steps = d.get("step"), d.get("max_steps")\n training_phase = d.get("phase")\n break\n break\n if step is not None and max_steps is not None:\n print(f"Training: Step {step}/{max_steps} ({100 * step / max_steps:.1f}%)")\n if training_phase:\n print(f"Phase: {training_phase}")\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished: {status.status}")\n break\n time.sleep(10)\n\nassert status.status == "completed"\n" }, { "type": "markdown", - "source": "### 8. Validate Output Model and Deployment\n\nWith `deployment_config` configured, the platform will create a NIM deployment for the base model after training. The fine-tuned LoRA adapter is enabled by default and automatically served through the deployment. Check the model entity and deployment status.", - "source_html": "

8. Validate Output Model and Deployment

\n

With deployment_config configured, the platform will create a NIM deployment for the base model after training. The fine-tuned LoRA adapter is enabled by default and automatically served through the deployment. Check the model entity and deployment status.

\n" + "source": "### 8. Validate Output Model and Deployment\n\nWith the base model entity and LoRA adapter from training, create a NIM deployment with `lora_enabled=True` so the adapter is served alongside the base weights. Check the model entity and deployment status.", + "source_html": "

8. Validate Output Model and Deployment

\n

With the base model entity and LoRA adapter from training, create a NIM deployment with lora_enabled=True so the adapter is served alongside the base weights. Check the model entity and deployment status.

\n" }, { "type": "code", - "source": "model_entity = client.models.retrieve(workspace=\"default\", name=MODEL_NAME)\n# Clear verbose linear_layers list for cleaner output\nmodel_entity.spec.linear_layers = None\nprint(model_entity.model_dump_json(indent=2))\n\ndeployment_name = sanitize_name(\"sft-deploy\", job.spec.model)\ndeployment_status = client.inference.deployments.retrieve(name=deployment_name, workspace=\"default\")\nprint(f\"Deployment status: {deployment_status.status}\")", + "source": "deploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"lora-deploy-cfg-{deploy_suffix}\"\ndeployment_name = f\"lora-deploy-{deploy_suffix}\"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": MODEL_NAME,\n \"lora_enabled\": True,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n \"additional_args\": [\"--max-lora-rank\", \"32\"],\n },\n)\n\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=deployment_name,\n config=deployment_config.name,\n)\n\nmodel_entity = client.models.retrieve(workspace=\"default\", name=MODEL_NAME)\nif model_entity.spec:\n model_entity.spec.linear_layers = None\nprint(model_entity.model_dump_json(indent=2))\nprint(f\"Deployment status: {deployment.status}\")", "language": "python", - "source_html": "model_entity = client.models.retrieve(workspace="default", name=MODEL_NAME)\n# Clear verbose linear_layers list for cleaner output\nmodel_entity.spec.linear_layers = None\nprint(model_entity.model_dump_json(indent=2))\n\ndeployment_name = sanitize_name("sft-deploy", job.spec.model)\ndeployment_status = client.inference.deployments.retrieve(name=deployment_name, workspace="default")\nprint(f"Deployment status: {deployment_status.status}")\n" + "source_html": "deploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"lora-deploy-cfg-{deploy_suffix}"\ndeployment_name = f"lora-deploy-{deploy_suffix}"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n engine="vllm",\n model_spec={\n "model_namespace": "default",\n "model_name": MODEL_NAME,\n "lora_enabled": True,\n },\n executor_config={\n "gpu": 1,\n "image_name": "vllm/vllm-openai",\n "image_tag": "v0.22.1",\n "additional_args": ["--max-lora-rank", "32"],\n },\n)\n\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=deployment_name,\n config=deployment_config.name,\n)\n\nmodel_entity = client.models.retrieve(workspace="default", name=MODEL_NAME)\nif model_entity.spec:\n model_entity.spec.linear_layers = None\nprint(model_entity.model_dump_json(indent=2))\nprint(f"Deployment status: {deployment.status}")\n" }, { "type": "markdown", @@ -129,13 +131,13 @@ export default { cells: [ }, { "type": "code", - "source": "context = \"The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit.\"\nquestion = \"Who was the first person to walk on the Moon?\"\nmessages = [\n {\"role\": \"user\", \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\"}\n]\nresponse = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment_name,\n workspace=\"default\",\n body={\n \"model\": job.spec.output.name,\n \"messages\": messages,\n \"temperature\": 0,\n \"max_tokens\": 256,\n }\n)\nprint(\"=\" * 60)\nprint(\"MODEL INFERENCE\")\nprint(\"=\" * 60)\nprint(f\"Question: {question}\")\nprint(f\"Expected: Neil Armstrong\")\nprint(f\"Model output: {response['choices'][0]['message']['content']}\")", + "source": "context = \"The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit.\"\nquestion = \"Who was the first person to walk on the Moon?\"\nmessages = [\n {\"role\": \"user\", \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\"}\n]\nresponse = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment_name,\n workspace=\"default\",\n body={\n \"model\": OUTPUT_NAME,\n \"messages\": messages,\n \"temperature\": 0,\n \"max_tokens\": 256,\n }\n)\nprint(\"=\" * 60)\nprint(\"MODEL INFERENCE\")\nprint(\"=\" * 60)\nprint(f\"Question: {question}\")\nprint(f\"Expected: Neil Armstrong\")\nprint(f\"Model output: {response['choices'][0]['message']['content']}\")", "language": "python", - "source_html": "context = "The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit."\nquestion = "Who was the first person to walk on the Moon?"\nmessages = [\n {"role": "user", "content": f"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}"}\n]\nresponse = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment_name,\n workspace="default",\n body={\n "model": job.spec.output.name,\n "messages": messages,\n "temperature": 0,\n "max_tokens": 256,\n }\n)\nprint("=" * 60)\nprint("MODEL INFERENCE")\nprint("=" * 60)\nprint(f"Question: {question}")\nprint(f"Expected: Neil Armstrong")\nprint(f"Model output: {response['choices'][0]['message']['content']}")\n" + "source_html": "context = "The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit."\nquestion = "Who was the first person to walk on the Moon?"\nmessages = [\n {"role": "user", "content": f"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}"}\n]\nresponse = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment_name,\n workspace="default",\n body={\n "model": OUTPUT_NAME,\n "messages": messages,\n "temperature": 0,\n "max_tokens": 256,\n }\n)\nprint("=" * 60)\nprint("MODEL INFERENCE")\nprint("=" * 60)\nprint(f"Question: {question}")\nprint(f"Expected: Neil Armstrong")\nprint(f"Model output: {response['choices'][0]['message']['content']}")\n" }, { "type": "markdown", - "source": "## Conclusion\n\nYou have started a LoRA customization job, monitored it to completion, and evaluated the fine-tuned model. Use the `output.name` to access the model for further inference or evaluation.\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Try [Full SFT](./sft-customization-job) or [DPO](./dpo-customization-job) for other customization options", - "source_html": "

Conclusion

\n

You have started a LoRA customization job, monitored it to completion, and evaluated the fine-tuned model. Use the output.name to access the model for further inference or evaluation.

\n

Next Steps

\n\n" + "source": "## Conclusion\n\nYou have started a LoRA customization job, monitored it to completion, and evaluated the fine-tuned model. Use the `output.name` to access the model for further inference or evaluation.\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Try [Full SFT](./sft-customization-job) for other customization options", + "source_html": "

Conclusion

\n

You have started a LoRA customization job, monitored it to completion, and evaluated the fine-tuned model. Use the output.name to access the model for further inference or evaluation.

\n

Next Steps

\n\n" } ] }; diff --git a/docs/fern/components/notebooks/optimize-throughput.json b/docs/fern/components/notebooks/optimize-throughput.json index 17d520aa81..452ccf038c 100644 --- a/docs/fern/components/notebooks/optimize-throughput.json +++ b/docs/fern/components/notebooks/optimize-throughput.json @@ -7,8 +7,8 @@ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (PyPI wrapper: pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  4. \n
\n" }, { "type": "markdown", @@ -50,9 +50,9 @@ }, { "type": "code", - "source": "# Create fileset to store SFT training data\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SFT training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate training data is uploaded correctly\nprint(\"Training data:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", + "source": "# Create fileset to store SFT training data\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SFT training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=f\"{DATASET_PATH}/\", # Trailing slash uploads directory contents to fileset root\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate training data is uploaded correctly\nprint(\"Training data:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", "language": "python", - "source_html": "# Create fileset to store SFT training data\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SFT training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate training data is uploaded correctly\nprint("Training data:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" + "source_html": "# Create fileset to store SFT training data\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SFT training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=f"{DATASET_PATH}/", # Trailing slash uploads directory contents to fileset root\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate training data is uploaded correctly\nprint("Training data:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" }, { "type": "markdown", @@ -78,14 +78,14 @@ }, { "type": "markdown", - "source": "### 5. Create LoRA Job with Sequence Packing\nCreate a customization job with an inline target referencing the base model and dataset filesets created in previous steps.", - "source_html": "

5. Create LoRA Job with Sequence Packing

\n

Create a customization job with an inline target referencing the base model and dataset filesets created in previous steps.

\n" + "source": "### 5. Create LoRA Job with Sequence Packing\nCreate a LoRA customization job with **sequence packing** enabled via `AutomodelJobInput` (`batch.sequence_packing=True`).", + "source_html": "

5. Create LoRA Job with Sequence Packing

\n

Create a LoRA customization job with sequence packing enabled via AutomodelJobInput (batch.sequence_packing=True).

\n" }, { "type": "code", - "source": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n LoRaParamsParam,\n)\n\n# Enable sequence packing to improve throughput and GPU utilization\nSEQUENCE_PACKING_ENABLED = True\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\njob_spec = CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=4096,\n val_check_interval=0.1,\n micro_batch_size=1,\n sequence_packing=SEQUENCE_PACKING_ENABLED,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n)\n\njob_with_sequence_packing = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=job_spec\n)\n\nprint(f\"Job ID: {job_with_sequence_packing.name}\")\nprint(f\"Output model: {job_with_sequence_packing.spec.output.name}\")", + "source": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\nSEQUENCE_PACKING_ENABLED = True\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"packing-job-{job_suffix}\"\nPACK_OUTPUT_NAME = f\"packing-out-{job_suffix}\"\n\nspec = AutomodelJobInput(\n model=f\"default/{base_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"lora\",\n \"max_seq_length\": 4096,\n },\n schedule={\"epochs\": 1, \"val_check_interval\": 0.1},\n batch={\n \"global_batch_size\": 64,\n \"micro_batch_size\": 1,\n \"sequence_packing\": SEQUENCE_PACKING_ENABLED,\n },\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": PACK_OUTPUT_NAME},\n)\n\njob_with_sequence_packing = client.customization.automodel.jobs.create(\n spec=spec, workspace=\"default\", name=JOB_NAME\n)\n\nprint(f\"Submitted job: {job_with_sequence_packing.job.name}\")\nprint(f\"Output adapter: {PACK_OUTPUT_NAME}\")", "language": "python", - "source_html": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n LoRaParamsParam,\n)\n\n# Enable sequence packing to improve throughput and GPU utilization\nSEQUENCE_PACKING_ENABLED = True\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f"my-sft-job-{job_suffix}"\njob_spec = CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=4096,\n val_check_interval=0.1,\n micro_batch_size=1,\n sequence_packing=SEQUENCE_PACKING_ENABLED,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n)\n\njob_with_sequence_packing = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=job_spec\n)\n\nprint(f"Job ID: {job_with_sequence_packing.name}")\nprint(f"Output model: {job_with_sequence_packing.spec.output.name}")\n" + "source_html": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\nSEQUENCE_PACKING_ENABLED = True\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"packing-job-{job_suffix}"\nPACK_OUTPUT_NAME = f"packing-out-{job_suffix}"\n\nspec = AutomodelJobInput(\n model=f"default/{base_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "lora",\n "max_seq_length": 4096,\n },\n schedule={"epochs": 1, "val_check_interval": 0.1},\n batch={\n "global_batch_size": 64,\n "micro_batch_size": 1,\n "sequence_packing": SEQUENCE_PACKING_ENABLED,\n },\n optimizer={"learning_rate": 5e-5},\n parallelism={"num_gpus_per_node": 1},\n output={"name": PACK_OUTPUT_NAME},\n)\n\njob_with_sequence_packing = client.customization.automodel.jobs.create(\n spec=spec, workspace="default", name=JOB_NAME\n)\n\nprint(f"Submitted job: {job_with_sequence_packing.job.name}")\nprint(f"Output adapter: {PACK_OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -110,20 +110,20 @@ }, { "type": "code", - "source": "import time\nfrom typing import cast\nfrom IPython.display import clear_output\nfrom nemo_platform.types.shared import PlatformJobStatusResponse\n\n# Timeout set to 30 minutes to accommodate typical LoRA training duration for this dataset size.\n# Actual training time will vary based on hardware, model size, and dataset complexity.\nTIMEOUT_SECONDS = 30 * 60 # 30 minutes\nVAL_LOSS_KEY = \"val_loss\"\nTRAIN_LOSS_KEY = \"loss\"\n\n# ---------------------------------------------------------------------------\n# Job polling with live dashboard\n# ---------------------------------------------------------------------------\n\ndef wait_for_job(\n workspace: str,\n job_name: str,\n timeout: int = TIMEOUT_SECONDS,\n poll_interval: int = 10,\n val_loss_key: str = VAL_LOSS_KEY,\n train_loss_key: str = TRAIN_LOSS_KEY,\n) -> PlatformJobStatusResponse:\n \"\"\"\n Poll job status until completed, failed, cancelled, or timeout.\n Displays a live dashboard with loss curves and GPU metrics.\n\n Args:\n workspace: The workspace where the job is running.\n job_name: The name of the job to monitor.\n timeout: Maximum time to wait in seconds (default: 30 minutes).\n poll_interval: Time between status checks in seconds (default: 10).\n\n Returns:\n The final job status response.\n \"\"\"\n start_time = time.time()\n\n # Time-series accumulators required for plotting\n elapsed_mins: list[float] = []\n val_losses: list[float | None] = []\n train_losses: list[float | None] = []\n vram_history: list[list[float]] = []\n util_history: list[list[float]] = []\n\n while True:\n elapsed = time.time() - start_time\n elapsed_min = elapsed / 60\n\n # Check for timeout\n if elapsed > timeout:\n error_message = f\"Timeout reached after {elapsed_min:.1f} minutes\"\n print(f\"\\n{error_message}\")\n print(\"Job did not complete within the timeout period.\")\n raise Exception(error_message)\n\n status = client.jobs.get_status(name=job_name, workspace=workspace)\n\n # -- Extract training progress from nested steps structure --\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n val_loss: float | None = None\n train_loss: float | None = None\n current_step_name: str | None = None\n current_step_phase: str | None = None\n\n for job_step in status.steps or []:\n # Track the current active step name and phase for progress display\n if job_step.tasks:\n task = job_step.tasks[0]\n td = task.status_details or {}\n phase = cast(str, td.get(\"phase\", \"\"))\n # Update current step if it's active or pending (not completed)\n if job_step.status in (\"active\", \"pending\"):\n current_step_name = job_step.name\n current_step_phase = phase or \"started\"\n\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n td = task.status_details or {}\n step = cast(int, td[\"step\"]) if \"step\" in td else None\n max_steps = cast(int, td[\"max_steps\"]) if \"max_steps\" in td else None\n training_phase = cast(str, td[\"phase\"]) if \"phase\" in td else None\n val_loss = float(td[val_loss_key]) if val_loss_key in td else None\n train_loss = float(td[train_loss_key]) if train_loss_key in td else None\n break\n break\n\n # Fall back to top-level status_details\n if status.status_details:\n if val_loss is None and val_loss_key in status.status_details:\n val_loss = float(status.status_details[val_loss_key])\n if train_loss is None and train_loss_key in status.status_details:\n train_loss = float(status.status_details[train_loss_key])\n\n # -- Collect GPU snapshot --\n vram_pcts, util_pcts = _get_gpu_snapshot()\n\n # -- Append to accumulators used for the plots --\n elapsed_mins.append(elapsed_min)\n val_losses.append(val_loss)\n train_losses.append(train_loss)\n vram_history.append(vram_pcts)\n util_history.append(util_pcts)\n\n # -- Build status strings --\n status_str = f\"Status: {status.status}\"\n if step is not None and max_steps is not None:\n pct = step / max_steps * 100\n step_str = f\"Step {step}/{max_steps} ({pct:.0f}%)\"\n if training_phase:\n step_str += f\" - {training_phase}\"\n else:\n if current_step_name and current_step_phase:\n step_str = f\"{current_step_name} - {current_step_phase}\"\n elif current_step_name:\n step_str = f\"{current_step_name}\"\n else:\n step_str = \"Waiting for training to start...\"\n elapsed_str = f\"Elapsed: {elapsed_min:.1f} min\"\n\n # -- Redraw dashboard --\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n\n # -- Check terminal conditions --\n if status.status.lower() == \"completed\":\n # Redraw dashboard one final time with \"completed\" status\n status_str = f\"Status: {status.status}\"\n if step is not None and max_steps is not None:\n step_str = f\"Step {max_steps}/{max_steps} (100%)\"\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n print(f\"\\nJob completed in {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n return status\n elif status.status.lower() in (\"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n print(f\"Total time elapsed: {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n\n # Print error details from the job level\n if status.error_details:\n error_msg = status.error_details.get(\"message\", \"\")\n if error_msg:\n print(f\"\\nError: {error_msg}\")\n\n # Find and print error details from the failed step/task\n for job_step in status.steps or []:\n if job_step.status == \"error\":\n print(f\"\\nFailed step: {job_step.name}\")\n if job_step.error_details:\n step_error = job_step.error_details.get(\"message\", \"\")\n if step_error:\n print(f\"Step error: {step_error}\")\n # Get error_stack from the failed task\n for task in job_step.tasks or []:\n if task.status == \"error\" and hasattr(task, \"error_stack\") and task.error_stack:\n print(f\"\\nError stack trace:\\n{task.error_stack}\")\n elif task.status == \"error\" and task.error_details:\n task_error = task.error_details.get(\"message\", \"\")\n if task_error:\n print(f\"Task error: {task_error}\")\n break\n\n raise Exception(f\"Job finished with status: {status.status}\")\n\n time.sleep(poll_interval)\n\n\n# Wait for the job to complete\njob_with_sequence_packing_status = wait_for_job(\n workspace=\"default\",\n job_name=job_with_sequence_packing.name,\n timeout=TIMEOUT_SECONDS,\n)\n\nprint(f\"Validation loss: {job_with_sequence_packing_status.status_details['val_loss']:.2f}\")", + "source": "import time\nfrom typing import cast\nfrom IPython.display import clear_output\nfrom nemo_platform.types.shared import PlatformJobStatusResponse\n\n# Timeout set to 30 minutes to accommodate typical LoRA training duration for this dataset size.\n# Actual training time will vary based on hardware, model size, and dataset complexity.\nTIMEOUT_SECONDS = 30 * 60 # 30 minutes\nVAL_LOSS_KEY = \"val_loss\"\nTRAIN_LOSS_KEY = \"loss\"\n\n# ---------------------------------------------------------------------------\n# Job polling with live dashboard\n# ---------------------------------------------------------------------------\n\ndef wait_for_job(\n workspace: str,\n job_name: str,\n timeout: int = TIMEOUT_SECONDS,\n poll_interval: int = 10,\n val_loss_key: str = VAL_LOSS_KEY,\n train_loss_key: str = TRAIN_LOSS_KEY,\n) -> PlatformJobStatusResponse:\n \"\"\"\n Poll job status until completed, failed, cancelled, or timeout.\n Displays a live dashboard with loss curves and GPU metrics.\n\n Args:\n workspace: The workspace where the job is running.\n job_name: The name of the job to monitor.\n timeout: Maximum time to wait in seconds (default: 30 minutes).\n poll_interval: Time between status checks in seconds (default: 10).\n\n Returns:\n The final job status response.\n \"\"\"\n start_time = time.time()\n\n # Time-series accumulators required for plotting\n elapsed_mins: list[float] = []\n val_losses: list[float | None] = []\n train_losses: list[float | None] = []\n vram_history: list[list[float]] = []\n util_history: list[list[float]] = []\n\n while True:\n elapsed = time.time() - start_time\n elapsed_min = elapsed / 60\n\n # Check for timeout\n if elapsed > timeout:\n error_message = f\"Timeout reached after {elapsed_min:.1f} minutes\"\n print(f\"\\n{error_message}\")\n print(\"Job did not complete within the timeout period.\")\n raise Exception(error_message)\n\n status = client.jobs.get_status(name=job_name, workspace=workspace)\n\n # -- Extract training progress from nested steps structure --\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n val_loss: float | None = None\n train_loss: float | None = None\n current_step_name: str | None = None\n current_step_phase: str | None = None\n\n for job_step in status.steps or []:\n # Track the current active step name and phase for progress display\n if job_step.tasks:\n task = job_step.tasks[0]\n td = task.status_details or {}\n phase = cast(str, td.get(\"phase\", \"\"))\n # Update current step if it's active or pending (not completed)\n if job_step.status in (\"active\", \"pending\"):\n current_step_name = job_step.name\n current_step_phase = phase or \"started\"\n\n if job_step.name == \"training\":\n for task in job_step.tasks or []:\n td = task.status_details or {}\n step = cast(int, td[\"step\"]) if \"step\" in td else None\n max_steps = cast(int, td[\"max_steps\"]) if \"max_steps\" in td else None\n training_phase = cast(str, td[\"phase\"]) if \"phase\" in td else None\n raw_val_loss = td.get(val_loss_key)\n val_loss = float(raw_val_loss) if raw_val_loss is not None else None\n raw_train_loss = td.get(train_loss_key)\n train_loss = float(raw_train_loss) if raw_train_loss is not None else None\n break\n break\n\n if val_loss is None:\n raw_val_loss = (status.status_details or {}).get(val_loss_key)\n val_loss = float(raw_val_loss) if raw_val_loss is not None else None\n if train_loss is None:\n raw_train_loss = (status.status_details or {}).get(train_loss_key)\n train_loss = float(raw_train_loss) if raw_train_loss is not None else None\n\n # -- Collect GPU snapshot --\n vram_pcts, util_pcts = _get_gpu_snapshot()\n\n # -- Append to accumulators used for the plots --\n elapsed_mins.append(elapsed_min)\n val_losses.append(val_loss)\n train_losses.append(train_loss)\n vram_history.append(vram_pcts)\n util_history.append(util_pcts)\n\n # -- Build status strings --\n status_str = f\"Status: {status.status}\"\n if step is not None and max_steps is not None:\n pct = step / max_steps * 100\n step_str = f\"Step {step}/{max_steps} ({pct:.0f}%)\"\n if training_phase:\n step_str += f\" - {training_phase}\"\n else:\n if current_step_name and current_step_phase:\n step_str = f\"{current_step_name} - {current_step_phase}\"\n elif current_step_name:\n step_str = f\"{current_step_name}\"\n else:\n step_str = \"Waiting for training to start...\"\n elapsed_str = f\"Elapsed: {elapsed_min:.1f} min\"\n\n # -- Redraw dashboard --\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n\n # -- Check terminal conditions --\n if status.status.lower() == \"completed\":\n # Redraw dashboard one final time with \"completed\" status\n status_str = f\"Status: {status.status}\"\n if step is not None and max_steps is not None:\n step_str = f\"Step {max_steps}/{max_steps} (100%)\"\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n print(f\"\\nJob completed in {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n return status\n elif status.status.lower() in (\"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n print(f\"Total time elapsed: {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n\n # Print error details from the job level\n if status.error_details:\n error_msg = status.error_details.get(\"message\", \"\")\n if error_msg:\n print(f\"\\nError: {error_msg}\")\n\n # Find and print error details from the failed step/task\n for job_step in status.steps or []:\n if job_step.status == \"error\":\n print(f\"\\nFailed step: {job_step.name}\")\n if job_step.error_details:\n step_error = job_step.error_details.get(\"message\", \"\")\n if step_error:\n print(f\"Step error: {step_error}\")\n # Get error_stack from the failed task\n for task in job_step.tasks or []:\n if task.status == \"error\" and hasattr(task, \"error_stack\") and task.error_stack:\n print(f\"\\nError stack trace:\\n{task.error_stack}\")\n elif task.status == \"error\" and task.error_details:\n task_error = task.error_details.get(\"message\", \"\")\n if task_error:\n print(f\"Task error: {task_error}\")\n break\n\n raise Exception(f\"Job finished with status: {status.status}\")\n\n time.sleep(poll_interval)\n\n\n# Wait for the job to complete\njob_with_sequence_packing_status = wait_for_job(\n workspace=\"default\",\n job_name=job_with_sequence_packing.job.name,\n timeout=TIMEOUT_SECONDS,\n)\n\npacked_val_loss = (job_with_sequence_packing_status.status_details or {}).get(\"val_loss\")\nif packed_val_loss is not None:\n print(f\"Validation loss: {float(packed_val_loss):.2f}\")\nelse:\n print(\"Validation loss: not reported in job status\")", "language": "python", - "source_html": "import time\nfrom typing import cast\nfrom IPython.display import clear_output\nfrom nemo_platform.types.shared import PlatformJobStatusResponse\n\n# Timeout set to 30 minutes to accommodate typical LoRA training duration for this dataset size.\n# Actual training time will vary based on hardware, model size, and dataset complexity.\nTIMEOUT_SECONDS = 30 * 60 # 30 minutes\nVAL_LOSS_KEY = "val_loss"\nTRAIN_LOSS_KEY = "loss"\n\n# ---------------------------------------------------------------------------\n# Job polling with live dashboard\n# ---------------------------------------------------------------------------\n\ndef wait_for_job(\n workspace: str,\n job_name: str,\n timeout: int = TIMEOUT_SECONDS,\n poll_interval: int = 10,\n val_loss_key: str = VAL_LOSS_KEY,\n train_loss_key: str = TRAIN_LOSS_KEY,\n) -> PlatformJobStatusResponse:\n """\n Poll job status until completed, failed, cancelled, or timeout.\n Displays a live dashboard with loss curves and GPU metrics.\n\n Args:\n workspace: The workspace where the job is running.\n job_name: The name of the job to monitor.\n timeout: Maximum time to wait in seconds (default: 30 minutes).\n poll_interval: Time between status checks in seconds (default: 10).\n\n Returns:\n The final job status response.\n """\n start_time = time.time()\n\n # Time-series accumulators required for plotting\n elapsed_mins: list[float] = []\n val_losses: list[float | None] = []\n train_losses: list[float | None] = []\n vram_history: list[list[float]] = []\n util_history: list[list[float]] = []\n\n while True:\n elapsed = time.time() - start_time\n elapsed_min = elapsed / 60\n\n # Check for timeout\n if elapsed > timeout:\n error_message = f"Timeout reached after {elapsed_min:.1f} minutes"\n print(f"\\n{error_message}")\n print("Job did not complete within the timeout period.")\n raise Exception(error_message)\n\n status = client.jobs.get_status(name=job_name, workspace=workspace)\n\n # -- Extract training progress from nested steps structure --\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n val_loss: float | None = None\n train_loss: float | None = None\n current_step_name: str | None = None\n current_step_phase: str | None = None\n\n for job_step in status.steps or []:\n # Track the current active step name and phase for progress display\n if job_step.tasks:\n task = job_step.tasks[0]\n td = task.status_details or {}\n phase = cast(str, td.get("phase", ""))\n # Update current step if it's active or pending (not completed)\n if job_step.status in ("active", "pending"):\n current_step_name = job_step.name\n current_step_phase = phase or "started"\n\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n td = task.status_details or {}\n step = cast(int, td["step"]) if "step" in td else None\n max_steps = cast(int, td["max_steps"]) if "max_steps" in td else None\n training_phase = cast(str, td["phase"]) if "phase" in td else None\n val_loss = float(td[val_loss_key]) if val_loss_key in td else None\n train_loss = float(td[train_loss_key]) if train_loss_key in td else None\n break\n break\n\n # Fall back to top-level status_details\n if status.status_details:\n if val_loss is None and val_loss_key in status.status_details:\n val_loss = float(status.status_details[val_loss_key])\n if train_loss is None and train_loss_key in status.status_details:\n train_loss = float(status.status_details[train_loss_key])\n\n # -- Collect GPU snapshot --\n vram_pcts, util_pcts = _get_gpu_snapshot()\n\n # -- Append to accumulators used for the plots --\n elapsed_mins.append(elapsed_min)\n val_losses.append(val_loss)\n train_losses.append(train_loss)\n vram_history.append(vram_pcts)\n util_history.append(util_pcts)\n\n # -- Build status strings --\n status_str = f"Status: {status.status}"\n if step is not None and max_steps is not None:\n pct = step / max_steps * 100\n step_str = f"Step {step}/{max_steps} ({pct:.0f}%)"\n if training_phase:\n step_str += f" - {training_phase}"\n else:\n if current_step_name and current_step_phase:\n step_str = f"{current_step_name} - {current_step_phase}"\n elif current_step_name:\n step_str = f"{current_step_name}"\n else:\n step_str = "Waiting for training to start..."\n elapsed_str = f"Elapsed: {elapsed_min:.1f} min"\n\n # -- Redraw dashboard --\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n\n # -- Check terminal conditions --\n if status.status.lower() == "completed":\n # Redraw dashboard one final time with "completed" status\n status_str = f"Status: {status.status}"\n if step is not None and max_steps is not None:\n step_str = f"Step {max_steps}/{max_steps} (100%)"\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n print(f"\\nJob completed in {elapsed_min:.1f} minutes ({elapsed:.0f}s)")\n return status\n elif status.status.lower() in ("failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n print(f"Total time elapsed: {elapsed_min:.1f} minutes ({elapsed:.0f}s)")\n\n # Print error details from the job level\n if status.error_details:\n error_msg = status.error_details.get("message", "")\n if error_msg:\n print(f"\\nError: {error_msg}")\n\n # Find and print error details from the failed step/task\n for job_step in status.steps or []:\n if job_step.status == "error":\n print(f"\\nFailed step: {job_step.name}")\n if job_step.error_details:\n step_error = job_step.error_details.get("message", "")\n if step_error:\n print(f"Step error: {step_error}")\n # Get error_stack from the failed task\n for task in job_step.tasks or []:\n if task.status == "error" and hasattr(task, "error_stack") and task.error_stack:\n print(f"\\nError stack trace:\\n{task.error_stack}")\n elif task.status == "error" and task.error_details:\n task_error = task.error_details.get("message", "")\n if task_error:\n print(f"Task error: {task_error}")\n break\n\n raise Exception(f"Job finished with status: {status.status}")\n\n time.sleep(poll_interval)\n\n\n# Wait for the job to complete\njob_with_sequence_packing_status = wait_for_job(\n workspace="default",\n job_name=job_with_sequence_packing.name,\n timeout=TIMEOUT_SECONDS,\n)\n\nprint(f"Validation loss: {job_with_sequence_packing_status.status_details['val_loss']:.2f}")\n" + "source_html": "import time\nfrom typing import cast\nfrom IPython.display import clear_output\nfrom nemo_platform.types.shared import PlatformJobStatusResponse\n\n# Timeout set to 30 minutes to accommodate typical LoRA training duration for this dataset size.\n# Actual training time will vary based on hardware, model size, and dataset complexity.\nTIMEOUT_SECONDS = 30 * 60 # 30 minutes\nVAL_LOSS_KEY = "val_loss"\nTRAIN_LOSS_KEY = "loss"\n\n# ---------------------------------------------------------------------------\n# Job polling with live dashboard\n# ---------------------------------------------------------------------------\n\ndef wait_for_job(\n workspace: str,\n job_name: str,\n timeout: int = TIMEOUT_SECONDS,\n poll_interval: int = 10,\n val_loss_key: str = VAL_LOSS_KEY,\n train_loss_key: str = TRAIN_LOSS_KEY,\n) -> PlatformJobStatusResponse:\n """\n Poll job status until completed, failed, cancelled, or timeout.\n Displays a live dashboard with loss curves and GPU metrics.\n\n Args:\n workspace: The workspace where the job is running.\n job_name: The name of the job to monitor.\n timeout: Maximum time to wait in seconds (default: 30 minutes).\n poll_interval: Time between status checks in seconds (default: 10).\n\n Returns:\n The final job status response.\n """\n start_time = time.time()\n\n # Time-series accumulators required for plotting\n elapsed_mins: list[float] = []\n val_losses: list[float | None] = []\n train_losses: list[float | None] = []\n vram_history: list[list[float]] = []\n util_history: list[list[float]] = []\n\n while True:\n elapsed = time.time() - start_time\n elapsed_min = elapsed / 60\n\n # Check for timeout\n if elapsed > timeout:\n error_message = f"Timeout reached after {elapsed_min:.1f} minutes"\n print(f"\\n{error_message}")\n print("Job did not complete within the timeout period.")\n raise Exception(error_message)\n\n status = client.jobs.get_status(name=job_name, workspace=workspace)\n\n # -- Extract training progress from nested steps structure --\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n val_loss: float | None = None\n train_loss: float | None = None\n current_step_name: str | None = None\n current_step_phase: str | None = None\n\n for job_step in status.steps or []:\n # Track the current active step name and phase for progress display\n if job_step.tasks:\n task = job_step.tasks[0]\n td = task.status_details or {}\n phase = cast(str, td.get("phase", ""))\n # Update current step if it's active or pending (not completed)\n if job_step.status in ("active", "pending"):\n current_step_name = job_step.name\n current_step_phase = phase or "started"\n\n if job_step.name == "training":\n for task in job_step.tasks or []:\n td = task.status_details or {}\n step = cast(int, td["step"]) if "step" in td else None\n max_steps = cast(int, td["max_steps"]) if "max_steps" in td else None\n training_phase = cast(str, td["phase"]) if "phase" in td else None\n raw_val_loss = td.get(val_loss_key)\n val_loss = float(raw_val_loss) if raw_val_loss is not None else None\n raw_train_loss = td.get(train_loss_key)\n train_loss = float(raw_train_loss) if raw_train_loss is not None else None\n break\n break\n\n if val_loss is None:\n raw_val_loss = (status.status_details or {}).get(val_loss_key)\n val_loss = float(raw_val_loss) if raw_val_loss is not None else None\n if train_loss is None:\n raw_train_loss = (status.status_details or {}).get(train_loss_key)\n train_loss = float(raw_train_loss) if raw_train_loss is not None else None\n\n # -- Collect GPU snapshot --\n vram_pcts, util_pcts = _get_gpu_snapshot()\n\n # -- Append to accumulators used for the plots --\n elapsed_mins.append(elapsed_min)\n val_losses.append(val_loss)\n train_losses.append(train_loss)\n vram_history.append(vram_pcts)\n util_history.append(util_pcts)\n\n # -- Build status strings --\n status_str = f"Status: {status.status}"\n if step is not None and max_steps is not None:\n pct = step / max_steps * 100\n step_str = f"Step {step}/{max_steps} ({pct:.0f}%)"\n if training_phase:\n step_str += f" - {training_phase}"\n else:\n if current_step_name and current_step_phase:\n step_str = f"{current_step_name} - {current_step_phase}"\n elif current_step_name:\n step_str = f"{current_step_name}"\n else:\n step_str = "Waiting for training to start..."\n elapsed_str = f"Elapsed: {elapsed_min:.1f} min"\n\n # -- Redraw dashboard --\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n\n # -- Check terminal conditions --\n if status.status.lower() == "completed":\n # Redraw dashboard one final time with "completed" status\n status_str = f"Status: {status.status}"\n if step is not None and max_steps is not None:\n step_str = f"Step {max_steps}/{max_steps} (100%)"\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n print(f"\\nJob completed in {elapsed_min:.1f} minutes ({elapsed:.0f}s)")\n return status\n elif status.status.lower() in ("failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n print(f"Total time elapsed: {elapsed_min:.1f} minutes ({elapsed:.0f}s)")\n\n # Print error details from the job level\n if status.error_details:\n error_msg = status.error_details.get("message", "")\n if error_msg:\n print(f"\\nError: {error_msg}")\n\n # Find and print error details from the failed step/task\n for job_step in status.steps or []:\n if job_step.status == "error":\n print(f"\\nFailed step: {job_step.name}")\n if job_step.error_details:\n step_error = job_step.error_details.get("message", "")\n if step_error:\n print(f"Step error: {step_error}")\n # Get error_stack from the failed task\n for task in job_step.tasks or []:\n if task.status == "error" and hasattr(task, "error_stack") and task.error_stack:\n print(f"\\nError stack trace:\\n{task.error_stack}")\n elif task.status == "error" and task.error_details:\n task_error = task.error_details.get("message", "")\n if task_error:\n print(f"Task error: {task_error}")\n break\n\n raise Exception(f"Job finished with status: {status.status}")\n\n time.sleep(poll_interval)\n\n\n# Wait for the job to complete\njob_with_sequence_packing_status = wait_for_job(\n workspace="default",\n job_name=job_with_sequence_packing.job.name,\n timeout=TIMEOUT_SECONDS,\n)\n\npacked_val_loss = (job_with_sequence_packing_status.status_details or {}).get("val_loss")\nif packed_val_loss is not None:\n print(f"Validation loss: {float(packed_val_loss):.2f}")\nelse:\n print("Validation loss: not reported in job status")\n" }, { "type": "markdown", - "source": "### 7. Create LoRA Job without Sequence Packing\nCreate a customization job with sequence packing disabled. It's expected to take longer to complete.", - "source_html": "

7. Create LoRA Job without Sequence Packing

\n

Create a customization job with sequence packing disabled. It's expected to take longer to complete.

\n" + "source": "### 7. Create LoRA Job without Sequence Packing\nCreate a second Automodel LoRA job with `batch.sequence_packing=False` for comparison.", + "source_html": "

7. Create LoRA Job without Sequence Packing

\n

Create a second Automodel LoRA job with batch.sequence_packing=False for comparison.

\n" }, { "type": "code", - "source": "import uuid\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\n\njob_spec_no_packing = CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=4096,\n val_check_interval=0.1,\n micro_batch_size=1,\n sequence_packing=False,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n)\n\njob_without_sequence_packing = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=job_spec_no_packing\n)\n\nprint(f\"Job ID: {job_without_sequence_packing.name}\")\nprint(f\"Output model: {job_without_sequence_packing.spec.output.name}\")", + "source": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"no-packing-job-{job_suffix}\"\nNO_PACK_OUTPUT_NAME = f\"no-packing-out-{job_suffix}\"\n\nspec = AutomodelJobInput(\n model=f\"default/{base_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"lora\",\n \"max_seq_length\": 4096,\n },\n schedule={\"epochs\": 1, \"val_check_interval\": 0.1},\n batch={\n \"global_batch_size\": 64,\n \"micro_batch_size\": 1,\n \"sequence_packing\": False,\n },\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": NO_PACK_OUTPUT_NAME},\n)\n\njob_without_sequence_packing = client.customization.automodel.jobs.create(\n spec=spec, workspace=\"default\", name=JOB_NAME\n)\n\nprint(f\"Submitted job: {job_without_sequence_packing.job.name}\")\nprint(f\"Output adapter: {NO_PACK_OUTPUT_NAME}\")", "language": "python", - "source_html": "import uuid\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"my-sft-job-{job_suffix}"\n\njob_spec_no_packing = CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=4096,\n val_check_interval=0.1,\n micro_batch_size=1,\n sequence_packing=False,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n)\n\njob_without_sequence_packing = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=job_spec_no_packing\n)\n\nprint(f"Job ID: {job_without_sequence_packing.name}")\nprint(f"Output model: {job_without_sequence_packing.spec.output.name}")\n" + "source_html": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"no-packing-job-{job_suffix}"\nNO_PACK_OUTPUT_NAME = f"no-packing-out-{job_suffix}"\n\nspec = AutomodelJobInput(\n model=f"default/{base_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "lora",\n "max_seq_length": 4096,\n },\n schedule={"epochs": 1, "val_check_interval": 0.1},\n batch={\n "global_batch_size": 64,\n "micro_batch_size": 1,\n "sequence_packing": False,\n },\n optimizer={"learning_rate": 5e-5},\n parallelism={"num_gpus_per_node": 1},\n output={"name": NO_PACK_OUTPUT_NAME},\n)\n\njob_without_sequence_packing = client.customization.automodel.jobs.create(\n spec=spec, workspace="default", name=JOB_NAME\n)\n\nprint(f"Submitted job: {job_without_sequence_packing.job.name}")\nprint(f"Output adapter: {NO_PACK_OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -132,9 +132,9 @@ }, { "type": "code", - "source": "# Wait for the training step to complete\njob_without_sequence_packing_status = wait_for_job(\n workspace=\"default\",\n job_name=job_without_sequence_packing.name,\n timeout=TIMEOUT_SECONDS\n)\n\nprint(f\"Validation loss: {job_without_sequence_packing_status.status_details['val_loss']:.2f}\")", + "source": "# Wait for the training step to complete\njob_without_sequence_packing_status = wait_for_job(\n workspace=\"default\",\n job_name=job_without_sequence_packing.job.name,\n timeout=TIMEOUT_SECONDS\n)\n\nno_pack_val_loss = (job_without_sequence_packing_status.status_details or {}).get(\"val_loss\")\nif no_pack_val_loss is not None:\n print(f\"Validation loss: {float(no_pack_val_loss):.2f}\")\nelse:\n print(\"Validation loss: not reported in job status\")", "language": "python", - "source_html": "# Wait for the training step to complete\njob_without_sequence_packing_status = wait_for_job(\n workspace="default",\n job_name=job_without_sequence_packing.name,\n timeout=TIMEOUT_SECONDS\n)\n\nprint(f"Validation loss: {job_without_sequence_packing_status.status_details['val_loss']:.2f}")\n" + "source_html": "# Wait for the training step to complete\njob_without_sequence_packing_status = wait_for_job(\n workspace="default",\n job_name=job_without_sequence_packing.job.name,\n timeout=TIMEOUT_SECONDS\n)\n\nno_pack_val_loss = (job_without_sequence_packing_status.status_details or {}).get("val_loss")\nif no_pack_val_loss is not None:\n print(f"Validation loss: {float(no_pack_val_loss):.2f}")\nelse:\n print("Validation loss: not reported in job status")\n" }, { "type": "markdown", @@ -143,9 +143,9 @@ }, { "type": "code", - "source": "from nemo_platform.types.jobs import PlatformJobStep\nfrom datetime import datetime\nimport pandas as pd\n\nSTEP_NAME = \"customization-training-job\"\n\ndef get_elapsed_time(step: PlatformJobStep) -> float:\n \"\"\"Calculate elapsed time in seconds from step's created_at to updated_at.\"\"\"\n created_at = datetime.fromisoformat(step.created_at.replace(\"Z\", \"+00:00\"))\n updated_at = datetime.fromisoformat(step.updated_at.replace(\"Z\", \"+00:00\"))\n return (updated_at - created_at).total_seconds()\n\nstep_with_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace=\"default\",\n job=job_with_sequence_packing.name,\n)\n\nstep_without_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace=\"default\",\n job=job_without_sequence_packing.name,\n)\n\ntime_to_complete_with_sequence_packing = get_elapsed_time(step_with_sequence_packing)\ntime_to_complete_without_sequence_packing = get_elapsed_time(step_without_sequence_packing)\n\n# Display results as a table\nresults_df = pd.DataFrame({\n \"Seq Packing Enabled\": [True, False],\n \"Val Loss\": [\n job_with_sequence_packing_status.status_details['val_loss'],\n job_without_sequence_packing_status.status_details['val_loss']\n ],\n \"Training Step Time, sec\": [\n time_to_complete_with_sequence_packing,\n time_to_complete_without_sequence_packing\n ]\n})\n\nresults_df.style.format({\"Val Loss\": \"{:.2f}\", \"Training Step Time, sec\": \"{:.0f}\"}).hide(axis='index')", + "source": "from nemo_platform.types.jobs import PlatformJobStep\nfrom datetime import datetime\nimport pandas as pd\n\nSTEP_NAME = \"training\"\n\ndef get_elapsed_time(step: PlatformJobStep) -> float:\n \"\"\"Calculate elapsed time in seconds from step's created_at to updated_at.\"\"\"\n created_at = datetime.fromisoformat(step.created_at.replace(\"Z\", \"+00:00\"))\n updated_at = datetime.fromisoformat(step.updated_at.replace(\"Z\", \"+00:00\"))\n return (updated_at - created_at).total_seconds()\n\nstep_with_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace=\"default\",\n job=job_with_sequence_packing.job.name,\n)\n\nstep_without_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace=\"default\",\n job=job_without_sequence_packing.job.name,\n)\n\ntime_to_complete_with_sequence_packing = get_elapsed_time(step_with_sequence_packing)\ntime_to_complete_without_sequence_packing = get_elapsed_time(step_without_sequence_packing)\n\n# Display results as a table\nresults_df = pd.DataFrame({\n \"Seq Packing Enabled\": [True, False],\n \"Val Loss\": [\n (job_with_sequence_packing_status.status_details or {}).get(\"val_loss\"),\n (job_without_sequence_packing_status.status_details or {}).get(\"val_loss\"),\n ],\n \"Training Step Time, sec\": [\n time_to_complete_with_sequence_packing,\n time_to_complete_without_sequence_packing\n ]\n})\n\nresults_df.style.format({\"Val Loss\": \"{:.2f}\", \"Training Step Time, sec\": \"{:.0f}\"}).hide(axis='index')", "language": "python", - "source_html": "from nemo_platform.types.jobs import PlatformJobStep\nfrom datetime import datetime\nimport pandas as pd\n\nSTEP_NAME = "customization-training-job"\n\ndef get_elapsed_time(step: PlatformJobStep) -> float:\n """Calculate elapsed time in seconds from step's created_at to updated_at."""\n created_at = datetime.fromisoformat(step.created_at.replace("Z", "+00:00"))\n updated_at = datetime.fromisoformat(step.updated_at.replace("Z", "+00:00"))\n return (updated_at - created_at).total_seconds()\n\nstep_with_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace="default",\n job=job_with_sequence_packing.name,\n)\n\nstep_without_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace="default",\n job=job_without_sequence_packing.name,\n)\n\ntime_to_complete_with_sequence_packing = get_elapsed_time(step_with_sequence_packing)\ntime_to_complete_without_sequence_packing = get_elapsed_time(step_without_sequence_packing)\n\n# Display results as a table\nresults_df = pd.DataFrame({\n "Seq Packing Enabled": [True, False],\n "Val Loss": [\n job_with_sequence_packing_status.status_details['val_loss'],\n job_without_sequence_packing_status.status_details['val_loss']\n ],\n "Training Step Time, sec": [\n time_to_complete_with_sequence_packing,\n time_to_complete_without_sequence_packing\n ]\n})\n\nresults_df.style.format({"Val Loss": "{:.2f}", "Training Step Time, sec": "{:.0f}"}).hide(axis='index')\n" + "source_html": "from nemo_platform.types.jobs import PlatformJobStep\nfrom datetime import datetime\nimport pandas as pd\n\nSTEP_NAME = "training"\n\ndef get_elapsed_time(step: PlatformJobStep) -> float:\n """Calculate elapsed time in seconds from step's created_at to updated_at."""\n created_at = datetime.fromisoformat(step.created_at.replace("Z", "+00:00"))\n updated_at = datetime.fromisoformat(step.updated_at.replace("Z", "+00:00"))\n return (updated_at - created_at).total_seconds()\n\nstep_with_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace="default",\n job=job_with_sequence_packing.job.name,\n)\n\nstep_without_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace="default",\n job=job_without_sequence_packing.job.name,\n)\n\ntime_to_complete_with_sequence_packing = get_elapsed_time(step_with_sequence_packing)\ntime_to_complete_without_sequence_packing = get_elapsed_time(step_without_sequence_packing)\n\n# Display results as a table\nresults_df = pd.DataFrame({\n "Seq Packing Enabled": [True, False],\n "Val Loss": [\n (job_with_sequence_packing_status.status_details or {}).get("val_loss"),\n (job_without_sequence_packing_status.status_details or {}).get("val_loss"),\n ],\n "Training Step Time, sec": [\n time_to_complete_with_sequence_packing,\n time_to_complete_without_sequence_packing\n ]\n})\n\nresults_df.style.format({"Val Loss": "{:.2f}", "Training Step Time, sec": "{:.0f}"}).hide(axis='index')\n" }, { "type": "markdown", diff --git a/docs/fern/components/notebooks/optimize-throughput.ts b/docs/fern/components/notebooks/optimize-throughput.ts index f25f2d527f..59211fc597 100644 --- a/docs/fern/components/notebooks/optimize-throughput.ts +++ b/docs/fern/components/notebooks/optimize-throughput.ts @@ -1,7 +1,9 @@ -// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -// SPDX-License-Identifier: Apache-2.0 - -/** Auto-generated by ipynb-to-fern-json.py - do not edit */ +/** + * SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + * + * Auto-generated by ipynb-to-fern-json.py - do not edit manually. + */ export default { cells: [ { "type": "markdown", @@ -10,8 +12,8 @@ export default { cells: [ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (PyPI wrapper: pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  4. \n
\n" }, { "type": "markdown", @@ -53,9 +55,9 @@ export default { cells: [ }, { "type": "code", - "source": "# Create fileset to store SFT training data\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SFT training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate training data is uploaded correctly\nprint(\"Training data:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", + "source": "# Create fileset to store SFT training data\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SFT training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=f\"{DATASET_PATH}/\", # Trailing slash uploads directory contents to fileset root\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate training data is uploaded correctly\nprint(\"Training data:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", "language": "python", - "source_html": "# Create fileset to store SFT training data\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SFT training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate training data is uploaded correctly\nprint("Training data:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" + "source_html": "# Create fileset to store SFT training data\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SFT training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=f"{DATASET_PATH}/", # Trailing slash uploads directory contents to fileset root\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate training data is uploaded correctly\nprint("Training data:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" }, { "type": "markdown", @@ -81,14 +83,14 @@ export default { cells: [ }, { "type": "markdown", - "source": "### 5. Create LoRA Job with Sequence Packing\nCreate a customization job with an inline target referencing the base model and dataset filesets created in previous steps.", - "source_html": "

5. Create LoRA Job with Sequence Packing

\n

Create a customization job with an inline target referencing the base model and dataset filesets created in previous steps.

\n" + "source": "### 5. Create LoRA Job with Sequence Packing\nCreate a LoRA customization job with **sequence packing** enabled via `AutomodelJobInput` (`batch.sequence_packing=True`).", + "source_html": "

5. Create LoRA Job with Sequence Packing

\n

Create a LoRA customization job with sequence packing enabled via AutomodelJobInput (batch.sequence_packing=True).

\n" }, { "type": "code", - "source": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n LoRaParamsParam,\n)\n\n# Enable sequence packing to improve throughput and GPU utilization\nSEQUENCE_PACKING_ENABLED = True\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\njob_spec = CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=4096,\n val_check_interval=0.1,\n micro_batch_size=1,\n sequence_packing=SEQUENCE_PACKING_ENABLED,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n)\n\njob_with_sequence_packing = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=job_spec\n)\n\nprint(f\"Job ID: {job_with_sequence_packing.name}\")\nprint(f\"Output model: {job_with_sequence_packing.spec.output.name}\")", + "source": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\nSEQUENCE_PACKING_ENABLED = True\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"packing-job-{job_suffix}\"\nPACK_OUTPUT_NAME = f\"packing-out-{job_suffix}\"\n\nspec = AutomodelJobInput(\n model=f\"default/{base_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"lora\",\n \"max_seq_length\": 4096,\n },\n schedule={\"epochs\": 1, \"val_check_interval\": 0.1},\n batch={\n \"global_batch_size\": 64,\n \"micro_batch_size\": 1,\n \"sequence_packing\": SEQUENCE_PACKING_ENABLED,\n },\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": PACK_OUTPUT_NAME},\n)\n\njob_with_sequence_packing = client.customization.automodel.jobs.create(\n spec=spec, workspace=\"default\", name=JOB_NAME\n)\n\nprint(f\"Submitted job: {job_with_sequence_packing.job.name}\")\nprint(f\"Output adapter: {PACK_OUTPUT_NAME}\")", "language": "python", - "source_html": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n LoRaParamsParam,\n)\n\n# Enable sequence packing to improve throughput and GPU utilization\nSEQUENCE_PACKING_ENABLED = True\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f"my-sft-job-{job_suffix}"\njob_spec = CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=4096,\n val_check_interval=0.1,\n micro_batch_size=1,\n sequence_packing=SEQUENCE_PACKING_ENABLED,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n)\n\njob_with_sequence_packing = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=job_spec\n)\n\nprint(f"Job ID: {job_with_sequence_packing.name}")\nprint(f"Output model: {job_with_sequence_packing.spec.output.name}")\n" + "source_html": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\nSEQUENCE_PACKING_ENABLED = True\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"packing-job-{job_suffix}"\nPACK_OUTPUT_NAME = f"packing-out-{job_suffix}"\n\nspec = AutomodelJobInput(\n model=f"default/{base_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "lora",\n "max_seq_length": 4096,\n },\n schedule={"epochs": 1, "val_check_interval": 0.1},\n batch={\n "global_batch_size": 64,\n "micro_batch_size": 1,\n "sequence_packing": SEQUENCE_PACKING_ENABLED,\n },\n optimizer={"learning_rate": 5e-5},\n parallelism={"num_gpus_per_node": 1},\n output={"name": PACK_OUTPUT_NAME},\n)\n\njob_with_sequence_packing = client.customization.automodel.jobs.create(\n spec=spec, workspace="default", name=JOB_NAME\n)\n\nprint(f"Submitted job: {job_with_sequence_packing.job.name}")\nprint(f"Output adapter: {PACK_OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -113,20 +115,20 @@ export default { cells: [ }, { "type": "code", - "source": "import time\nfrom typing import cast\nfrom IPython.display import clear_output\nfrom nemo_platform.types.shared import PlatformJobStatusResponse\n\n# Timeout set to 30 minutes to accommodate typical LoRA training duration for this dataset size.\n# Actual training time will vary based on hardware, model size, and dataset complexity.\nTIMEOUT_SECONDS = 30 * 60 # 30 minutes\nVAL_LOSS_KEY = \"val_loss\"\nTRAIN_LOSS_KEY = \"loss\"\n\n# ---------------------------------------------------------------------------\n# Job polling with live dashboard\n# ---------------------------------------------------------------------------\n\ndef wait_for_job(\n workspace: str,\n job_name: str,\n timeout: int = TIMEOUT_SECONDS,\n poll_interval: int = 10,\n val_loss_key: str = VAL_LOSS_KEY,\n train_loss_key: str = TRAIN_LOSS_KEY,\n) -> PlatformJobStatusResponse:\n \"\"\"\n Poll job status until completed, failed, cancelled, or timeout.\n Displays a live dashboard with loss curves and GPU metrics.\n\n Args:\n workspace: The workspace where the job is running.\n job_name: The name of the job to monitor.\n timeout: Maximum time to wait in seconds (default: 30 minutes).\n poll_interval: Time between status checks in seconds (default: 10).\n\n Returns:\n The final job status response.\n \"\"\"\n start_time = time.time()\n\n # Time-series accumulators required for plotting\n elapsed_mins: list[float] = []\n val_losses: list[float | None] = []\n train_losses: list[float | None] = []\n vram_history: list[list[float]] = []\n util_history: list[list[float]] = []\n\n while True:\n elapsed = time.time() - start_time\n elapsed_min = elapsed / 60\n\n # Check for timeout\n if elapsed > timeout:\n error_message = f\"Timeout reached after {elapsed_min:.1f} minutes\"\n print(f\"\\n{error_message}\")\n print(\"Job did not complete within the timeout period.\")\n raise Exception(error_message)\n\n status = client.jobs.get_status(name=job_name, workspace=workspace)\n\n # -- Extract training progress from nested steps structure --\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n val_loss: float | None = None\n train_loss: float | None = None\n current_step_name: str | None = None\n current_step_phase: str | None = None\n\n for job_step in status.steps or []:\n # Track the current active step name and phase for progress display\n if job_step.tasks:\n task = job_step.tasks[0]\n td = task.status_details or {}\n phase = cast(str, td.get(\"phase\", \"\"))\n # Update current step if it's active or pending (not completed)\n if job_step.status in (\"active\", \"pending\"):\n current_step_name = job_step.name\n current_step_phase = phase or \"started\"\n\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n td = task.status_details or {}\n step = cast(int, td[\"step\"]) if \"step\" in td else None\n max_steps = cast(int, td[\"max_steps\"]) if \"max_steps\" in td else None\n training_phase = cast(str, td[\"phase\"]) if \"phase\" in td else None\n val_loss = float(td[val_loss_key]) if val_loss_key in td else None\n train_loss = float(td[train_loss_key]) if train_loss_key in td else None\n break\n break\n\n # Fall back to top-level status_details\n if status.status_details:\n if val_loss is None and val_loss_key in status.status_details:\n val_loss = float(status.status_details[val_loss_key])\n if train_loss is None and train_loss_key in status.status_details:\n train_loss = float(status.status_details[train_loss_key])\n\n # -- Collect GPU snapshot --\n vram_pcts, util_pcts = _get_gpu_snapshot()\n\n # -- Append to accumulators used for the plots --\n elapsed_mins.append(elapsed_min)\n val_losses.append(val_loss)\n train_losses.append(train_loss)\n vram_history.append(vram_pcts)\n util_history.append(util_pcts)\n\n # -- Build status strings --\n status_str = f\"Status: {status.status}\"\n if step is not None and max_steps is not None:\n pct = step / max_steps * 100\n step_str = f\"Step {step}/{max_steps} ({pct:.0f}%)\"\n if training_phase:\n step_str += f\" - {training_phase}\"\n else:\n if current_step_name and current_step_phase:\n step_str = f\"{current_step_name} - {current_step_phase}\"\n elif current_step_name:\n step_str = f\"{current_step_name}\"\n else:\n step_str = \"Waiting for training to start...\"\n elapsed_str = f\"Elapsed: {elapsed_min:.1f} min\"\n\n # -- Redraw dashboard --\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n\n # -- Check terminal conditions --\n if status.status.lower() == \"completed\":\n # Redraw dashboard one final time with \"completed\" status\n status_str = f\"Status: {status.status}\"\n if step is not None and max_steps is not None:\n step_str = f\"Step {max_steps}/{max_steps} (100%)\"\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n print(f\"\\nJob completed in {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n return status\n elif status.status.lower() in (\"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n print(f\"Total time elapsed: {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n\n # Print error details from the job level\n if status.error_details:\n error_msg = status.error_details.get(\"message\", \"\")\n if error_msg:\n print(f\"\\nError: {error_msg}\")\n\n # Find and print error details from the failed step/task\n for job_step in status.steps or []:\n if job_step.status == \"error\":\n print(f\"\\nFailed step: {job_step.name}\")\n if job_step.error_details:\n step_error = job_step.error_details.get(\"message\", \"\")\n if step_error:\n print(f\"Step error: {step_error}\")\n # Get error_stack from the failed task\n for task in job_step.tasks or []:\n if task.status == \"error\" and hasattr(task, \"error_stack\") and task.error_stack:\n print(f\"\\nError stack trace:\\n{task.error_stack}\")\n elif task.status == \"error\" and task.error_details:\n task_error = task.error_details.get(\"message\", \"\")\n if task_error:\n print(f\"Task error: {task_error}\")\n break\n\n raise Exception(f\"Job finished with status: {status.status}\")\n\n time.sleep(poll_interval)\n\n\n# Wait for the job to complete\njob_with_sequence_packing_status = wait_for_job(\n workspace=\"default\",\n job_name=job_with_sequence_packing.name,\n timeout=TIMEOUT_SECONDS,\n)\n\nprint(f\"Validation loss: {job_with_sequence_packing_status.status_details['val_loss']:.2f}\")", + "source": "import time\nfrom typing import cast\nfrom IPython.display import clear_output\nfrom nemo_platform.types.shared import PlatformJobStatusResponse\n\n# Timeout set to 30 minutes to accommodate typical LoRA training duration for this dataset size.\n# Actual training time will vary based on hardware, model size, and dataset complexity.\nTIMEOUT_SECONDS = 30 * 60 # 30 minutes\nVAL_LOSS_KEY = \"val_loss\"\nTRAIN_LOSS_KEY = \"loss\"\n\n# ---------------------------------------------------------------------------\n# Job polling with live dashboard\n# ---------------------------------------------------------------------------\n\ndef wait_for_job(\n workspace: str,\n job_name: str,\n timeout: int = TIMEOUT_SECONDS,\n poll_interval: int = 10,\n val_loss_key: str = VAL_LOSS_KEY,\n train_loss_key: str = TRAIN_LOSS_KEY,\n) -> PlatformJobStatusResponse:\n \"\"\"\n Poll job status until completed, failed, cancelled, or timeout.\n Displays a live dashboard with loss curves and GPU metrics.\n\n Args:\n workspace: The workspace where the job is running.\n job_name: The name of the job to monitor.\n timeout: Maximum time to wait in seconds (default: 30 minutes).\n poll_interval: Time between status checks in seconds (default: 10).\n\n Returns:\n The final job status response.\n \"\"\"\n start_time = time.time()\n\n # Time-series accumulators required for plotting\n elapsed_mins: list[float] = []\n val_losses: list[float | None] = []\n train_losses: list[float | None] = []\n vram_history: list[list[float]] = []\n util_history: list[list[float]] = []\n\n while True:\n elapsed = time.time() - start_time\n elapsed_min = elapsed / 60\n\n # Check for timeout\n if elapsed > timeout:\n error_message = f\"Timeout reached after {elapsed_min:.1f} minutes\"\n print(f\"\\n{error_message}\")\n print(\"Job did not complete within the timeout period.\")\n raise Exception(error_message)\n\n status = client.jobs.get_status(name=job_name, workspace=workspace)\n\n # -- Extract training progress from nested steps structure --\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n val_loss: float | None = None\n train_loss: float | None = None\n current_step_name: str | None = None\n current_step_phase: str | None = None\n\n for job_step in status.steps or []:\n # Track the current active step name and phase for progress display\n if job_step.tasks:\n task = job_step.tasks[0]\n td = task.status_details or {}\n phase = cast(str, td.get(\"phase\", \"\"))\n # Update current step if it's active or pending (not completed)\n if job_step.status in (\"active\", \"pending\"):\n current_step_name = job_step.name\n current_step_phase = phase or \"started\"\n\n if job_step.name == \"training\":\n for task in job_step.tasks or []:\n td = task.status_details or {}\n step = cast(int, td[\"step\"]) if \"step\" in td else None\n max_steps = cast(int, td[\"max_steps\"]) if \"max_steps\" in td else None\n training_phase = cast(str, td[\"phase\"]) if \"phase\" in td else None\n raw_val_loss = td.get(val_loss_key)\n val_loss = float(raw_val_loss) if raw_val_loss is not None else None\n raw_train_loss = td.get(train_loss_key)\n train_loss = float(raw_train_loss) if raw_train_loss is not None else None\n break\n break\n\n if val_loss is None:\n raw_val_loss = (status.status_details or {}).get(val_loss_key)\n val_loss = float(raw_val_loss) if raw_val_loss is not None else None\n if train_loss is None:\n raw_train_loss = (status.status_details or {}).get(train_loss_key)\n train_loss = float(raw_train_loss) if raw_train_loss is not None else None\n\n # -- Collect GPU snapshot --\n vram_pcts, util_pcts = _get_gpu_snapshot()\n\n # -- Append to accumulators used for the plots --\n elapsed_mins.append(elapsed_min)\n val_losses.append(val_loss)\n train_losses.append(train_loss)\n vram_history.append(vram_pcts)\n util_history.append(util_pcts)\n\n # -- Build status strings --\n status_str = f\"Status: {status.status}\"\n if step is not None and max_steps is not None:\n pct = step / max_steps * 100\n step_str = f\"Step {step}/{max_steps} ({pct:.0f}%)\"\n if training_phase:\n step_str += f\" - {training_phase}\"\n else:\n if current_step_name and current_step_phase:\n step_str = f\"{current_step_name} - {current_step_phase}\"\n elif current_step_name:\n step_str = f\"{current_step_name}\"\n else:\n step_str = \"Waiting for training to start...\"\n elapsed_str = f\"Elapsed: {elapsed_min:.1f} min\"\n\n # -- Redraw dashboard --\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n\n # -- Check terminal conditions --\n if status.status.lower() == \"completed\":\n # Redraw dashboard one final time with \"completed\" status\n status_str = f\"Status: {status.status}\"\n if step is not None and max_steps is not None:\n step_str = f\"Step {max_steps}/{max_steps} (100%)\"\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n print(f\"\\nJob completed in {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n return status\n elif status.status.lower() in (\"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n print(f\"Total time elapsed: {elapsed_min:.1f} minutes ({elapsed:.0f}s)\")\n\n # Print error details from the job level\n if status.error_details:\n error_msg = status.error_details.get(\"message\", \"\")\n if error_msg:\n print(f\"\\nError: {error_msg}\")\n\n # Find and print error details from the failed step/task\n for job_step in status.steps or []:\n if job_step.status == \"error\":\n print(f\"\\nFailed step: {job_step.name}\")\n if job_step.error_details:\n step_error = job_step.error_details.get(\"message\", \"\")\n if step_error:\n print(f\"Step error: {step_error}\")\n # Get error_stack from the failed task\n for task in job_step.tasks or []:\n if task.status == \"error\" and hasattr(task, \"error_stack\") and task.error_stack:\n print(f\"\\nError stack trace:\\n{task.error_stack}\")\n elif task.status == \"error\" and task.error_details:\n task_error = task.error_details.get(\"message\", \"\")\n if task_error:\n print(f\"Task error: {task_error}\")\n break\n\n raise Exception(f\"Job finished with status: {status.status}\")\n\n time.sleep(poll_interval)\n\n\n# Wait for the job to complete\njob_with_sequence_packing_status = wait_for_job(\n workspace=\"default\",\n job_name=job_with_sequence_packing.job.name,\n timeout=TIMEOUT_SECONDS,\n)\n\npacked_val_loss = (job_with_sequence_packing_status.status_details or {}).get(\"val_loss\")\nif packed_val_loss is not None:\n print(f\"Validation loss: {float(packed_val_loss):.2f}\")\nelse:\n print(\"Validation loss: not reported in job status\")", "language": "python", - "source_html": "import time\nfrom typing import cast\nfrom IPython.display import clear_output\nfrom nemo_platform.types.shared import PlatformJobStatusResponse\n\n# Timeout set to 30 minutes to accommodate typical LoRA training duration for this dataset size.\n# Actual training time will vary based on hardware, model size, and dataset complexity.\nTIMEOUT_SECONDS = 30 * 60 # 30 minutes\nVAL_LOSS_KEY = "val_loss"\nTRAIN_LOSS_KEY = "loss"\n\n# ---------------------------------------------------------------------------\n# Job polling with live dashboard\n# ---------------------------------------------------------------------------\n\ndef wait_for_job(\n workspace: str,\n job_name: str,\n timeout: int = TIMEOUT_SECONDS,\n poll_interval: int = 10,\n val_loss_key: str = VAL_LOSS_KEY,\n train_loss_key: str = TRAIN_LOSS_KEY,\n) -> PlatformJobStatusResponse:\n """\n Poll job status until completed, failed, cancelled, or timeout.\n Displays a live dashboard with loss curves and GPU metrics.\n\n Args:\n workspace: The workspace where the job is running.\n job_name: The name of the job to monitor.\n timeout: Maximum time to wait in seconds (default: 30 minutes).\n poll_interval: Time between status checks in seconds (default: 10).\n\n Returns:\n The final job status response.\n """\n start_time = time.time()\n\n # Time-series accumulators required for plotting\n elapsed_mins: list[float] = []\n val_losses: list[float | None] = []\n train_losses: list[float | None] = []\n vram_history: list[list[float]] = []\n util_history: list[list[float]] = []\n\n while True:\n elapsed = time.time() - start_time\n elapsed_min = elapsed / 60\n\n # Check for timeout\n if elapsed > timeout:\n error_message = f"Timeout reached after {elapsed_min:.1f} minutes"\n print(f"\\n{error_message}")\n print("Job did not complete within the timeout period.")\n raise Exception(error_message)\n\n status = client.jobs.get_status(name=job_name, workspace=workspace)\n\n # -- Extract training progress from nested steps structure --\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n val_loss: float | None = None\n train_loss: float | None = None\n current_step_name: str | None = None\n current_step_phase: str | None = None\n\n for job_step in status.steps or []:\n # Track the current active step name and phase for progress display\n if job_step.tasks:\n task = job_step.tasks[0]\n td = task.status_details or {}\n phase = cast(str, td.get("phase", ""))\n # Update current step if it's active or pending (not completed)\n if job_step.status in ("active", "pending"):\n current_step_name = job_step.name\n current_step_phase = phase or "started"\n\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n td = task.status_details or {}\n step = cast(int, td["step"]) if "step" in td else None\n max_steps = cast(int, td["max_steps"]) if "max_steps" in td else None\n training_phase = cast(str, td["phase"]) if "phase" in td else None\n val_loss = float(td[val_loss_key]) if val_loss_key in td else None\n train_loss = float(td[train_loss_key]) if train_loss_key in td else None\n break\n break\n\n # Fall back to top-level status_details\n if status.status_details:\n if val_loss is None and val_loss_key in status.status_details:\n val_loss = float(status.status_details[val_loss_key])\n if train_loss is None and train_loss_key in status.status_details:\n train_loss = float(status.status_details[train_loss_key])\n\n # -- Collect GPU snapshot --\n vram_pcts, util_pcts = _get_gpu_snapshot()\n\n # -- Append to accumulators used for the plots --\n elapsed_mins.append(elapsed_min)\n val_losses.append(val_loss)\n train_losses.append(train_loss)\n vram_history.append(vram_pcts)\n util_history.append(util_pcts)\n\n # -- Build status strings --\n status_str = f"Status: {status.status}"\n if step is not None and max_steps is not None:\n pct = step / max_steps * 100\n step_str = f"Step {step}/{max_steps} ({pct:.0f}%)"\n if training_phase:\n step_str += f" - {training_phase}"\n else:\n if current_step_name and current_step_phase:\n step_str = f"{current_step_name} - {current_step_phase}"\n elif current_step_name:\n step_str = f"{current_step_name}"\n else:\n step_str = "Waiting for training to start..."\n elapsed_str = f"Elapsed: {elapsed_min:.1f} min"\n\n # -- Redraw dashboard --\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n\n # -- Check terminal conditions --\n if status.status.lower() == "completed":\n # Redraw dashboard one final time with "completed" status\n status_str = f"Status: {status.status}"\n if step is not None and max_steps is not None:\n step_str = f"Step {max_steps}/{max_steps} (100%)"\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n print(f"\\nJob completed in {elapsed_min:.1f} minutes ({elapsed:.0f}s)")\n return status\n elif status.status.lower() in ("failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n print(f"Total time elapsed: {elapsed_min:.1f} minutes ({elapsed:.0f}s)")\n\n # Print error details from the job level\n if status.error_details:\n error_msg = status.error_details.get("message", "")\n if error_msg:\n print(f"\\nError: {error_msg}")\n\n # Find and print error details from the failed step/task\n for job_step in status.steps or []:\n if job_step.status == "error":\n print(f"\\nFailed step: {job_step.name}")\n if job_step.error_details:\n step_error = job_step.error_details.get("message", "")\n if step_error:\n print(f"Step error: {step_error}")\n # Get error_stack from the failed task\n for task in job_step.tasks or []:\n if task.status == "error" and hasattr(task, "error_stack") and task.error_stack:\n print(f"\\nError stack trace:\\n{task.error_stack}")\n elif task.status == "error" and task.error_details:\n task_error = task.error_details.get("message", "")\n if task_error:\n print(f"Task error: {task_error}")\n break\n\n raise Exception(f"Job finished with status: {status.status}")\n\n time.sleep(poll_interval)\n\n\n# Wait for the job to complete\njob_with_sequence_packing_status = wait_for_job(\n workspace="default",\n job_name=job_with_sequence_packing.name,\n timeout=TIMEOUT_SECONDS,\n)\n\nprint(f"Validation loss: {job_with_sequence_packing_status.status_details['val_loss']:.2f}")\n" + "source_html": "import time\nfrom typing import cast\nfrom IPython.display import clear_output\nfrom nemo_platform.types.shared import PlatformJobStatusResponse\n\n# Timeout set to 30 minutes to accommodate typical LoRA training duration for this dataset size.\n# Actual training time will vary based on hardware, model size, and dataset complexity.\nTIMEOUT_SECONDS = 30 * 60 # 30 minutes\nVAL_LOSS_KEY = "val_loss"\nTRAIN_LOSS_KEY = "loss"\n\n# ---------------------------------------------------------------------------\n# Job polling with live dashboard\n# ---------------------------------------------------------------------------\n\ndef wait_for_job(\n workspace: str,\n job_name: str,\n timeout: int = TIMEOUT_SECONDS,\n poll_interval: int = 10,\n val_loss_key: str = VAL_LOSS_KEY,\n train_loss_key: str = TRAIN_LOSS_KEY,\n) -> PlatformJobStatusResponse:\n """\n Poll job status until completed, failed, cancelled, or timeout.\n Displays a live dashboard with loss curves and GPU metrics.\n\n Args:\n workspace: The workspace where the job is running.\n job_name: The name of the job to monitor.\n timeout: Maximum time to wait in seconds (default: 30 minutes).\n poll_interval: Time between status checks in seconds (default: 10).\n\n Returns:\n The final job status response.\n """\n start_time = time.time()\n\n # Time-series accumulators required for plotting\n elapsed_mins: list[float] = []\n val_losses: list[float | None] = []\n train_losses: list[float | None] = []\n vram_history: list[list[float]] = []\n util_history: list[list[float]] = []\n\n while True:\n elapsed = time.time() - start_time\n elapsed_min = elapsed / 60\n\n # Check for timeout\n if elapsed > timeout:\n error_message = f"Timeout reached after {elapsed_min:.1f} minutes"\n print(f"\\n{error_message}")\n print("Job did not complete within the timeout period.")\n raise Exception(error_message)\n\n status = client.jobs.get_status(name=job_name, workspace=workspace)\n\n # -- Extract training progress from nested steps structure --\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n val_loss: float | None = None\n train_loss: float | None = None\n current_step_name: str | None = None\n current_step_phase: str | None = None\n\n for job_step in status.steps or []:\n # Track the current active step name and phase for progress display\n if job_step.tasks:\n task = job_step.tasks[0]\n td = task.status_details or {}\n phase = cast(str, td.get("phase", ""))\n # Update current step if it's active or pending (not completed)\n if job_step.status in ("active", "pending"):\n current_step_name = job_step.name\n current_step_phase = phase or "started"\n\n if job_step.name == "training":\n for task in job_step.tasks or []:\n td = task.status_details or {}\n step = cast(int, td["step"]) if "step" in td else None\n max_steps = cast(int, td["max_steps"]) if "max_steps" in td else None\n training_phase = cast(str, td["phase"]) if "phase" in td else None\n raw_val_loss = td.get(val_loss_key)\n val_loss = float(raw_val_loss) if raw_val_loss is not None else None\n raw_train_loss = td.get(train_loss_key)\n train_loss = float(raw_train_loss) if raw_train_loss is not None else None\n break\n break\n\n if val_loss is None:\n raw_val_loss = (status.status_details or {}).get(val_loss_key)\n val_loss = float(raw_val_loss) if raw_val_loss is not None else None\n if train_loss is None:\n raw_train_loss = (status.status_details or {}).get(train_loss_key)\n train_loss = float(raw_train_loss) if raw_train_loss is not None else None\n\n # -- Collect GPU snapshot --\n vram_pcts, util_pcts = _get_gpu_snapshot()\n\n # -- Append to accumulators used for the plots --\n elapsed_mins.append(elapsed_min)\n val_losses.append(val_loss)\n train_losses.append(train_loss)\n vram_history.append(vram_pcts)\n util_history.append(util_pcts)\n\n # -- Build status strings --\n status_str = f"Status: {status.status}"\n if step is not None and max_steps is not None:\n pct = step / max_steps * 100\n step_str = f"Step {step}/{max_steps} ({pct:.0f}%)"\n if training_phase:\n step_str += f" - {training_phase}"\n else:\n if current_step_name and current_step_phase:\n step_str = f"{current_step_name} - {current_step_phase}"\n elif current_step_name:\n step_str = f"{current_step_name}"\n else:\n step_str = "Waiting for training to start..."\n elapsed_str = f"Elapsed: {elapsed_min:.1f} min"\n\n # -- Redraw dashboard --\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n\n # -- Check terminal conditions --\n if status.status.lower() == "completed":\n # Redraw dashboard one final time with "completed" status\n status_str = f"Status: {status.status}"\n if step is not None and max_steps is not None:\n step_str = f"Step {max_steps}/{max_steps} (100%)"\n clear_output(wait=True)\n _draw_dashboard(\n elapsed_mins, val_losses, train_losses,\n vram_history, util_history,\n job_name, status_str, step_str, elapsed_str,\n )\n print(f"\\nJob completed in {elapsed_min:.1f} minutes ({elapsed:.0f}s)")\n return status\n elif status.status.lower() in ("failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n print(f"Total time elapsed: {elapsed_min:.1f} minutes ({elapsed:.0f}s)")\n\n # Print error details from the job level\n if status.error_details:\n error_msg = status.error_details.get("message", "")\n if error_msg:\n print(f"\\nError: {error_msg}")\n\n # Find and print error details from the failed step/task\n for job_step in status.steps or []:\n if job_step.status == "error":\n print(f"\\nFailed step: {job_step.name}")\n if job_step.error_details:\n step_error = job_step.error_details.get("message", "")\n if step_error:\n print(f"Step error: {step_error}")\n # Get error_stack from the failed task\n for task in job_step.tasks or []:\n if task.status == "error" and hasattr(task, "error_stack") and task.error_stack:\n print(f"\\nError stack trace:\\n{task.error_stack}")\n elif task.status == "error" and task.error_details:\n task_error = task.error_details.get("message", "")\n if task_error:\n print(f"Task error: {task_error}")\n break\n\n raise Exception(f"Job finished with status: {status.status}")\n\n time.sleep(poll_interval)\n\n\n# Wait for the job to complete\njob_with_sequence_packing_status = wait_for_job(\n workspace="default",\n job_name=job_with_sequence_packing.job.name,\n timeout=TIMEOUT_SECONDS,\n)\n\npacked_val_loss = (job_with_sequence_packing_status.status_details or {}).get("val_loss")\nif packed_val_loss is not None:\n print(f"Validation loss: {float(packed_val_loss):.2f}")\nelse:\n print("Validation loss: not reported in job status")\n" }, { "type": "markdown", - "source": "### 7. Create LoRA Job without Sequence Packing\nCreate a customization job with sequence packing disabled. It's expected to take longer to complete.", - "source_html": "

7. Create LoRA Job without Sequence Packing

\n

Create a customization job with sequence packing disabled. It's expected to take longer to complete.

\n" + "source": "### 7. Create LoRA Job without Sequence Packing\nCreate a second Automodel LoRA job with `batch.sequence_packing=False` for comparison.", + "source_html": "

7. Create LoRA Job without Sequence Packing

\n

Create a second Automodel LoRA job with batch.sequence_packing=False for comparison.

\n" }, { "type": "code", - "source": "import uuid\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\n\njob_spec_no_packing = CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=4096,\n val_check_interval=0.1,\n micro_batch_size=1,\n sequence_packing=False,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n)\n\njob_without_sequence_packing = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=job_spec_no_packing\n)\n\nprint(f\"Job ID: {job_without_sequence_packing.name}\")\nprint(f\"Output model: {job_without_sequence_packing.spec.output.name}\")", + "source": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"no-packing-job-{job_suffix}\"\nNO_PACK_OUTPUT_NAME = f\"no-packing-out-{job_suffix}\"\n\nspec = AutomodelJobInput(\n model=f\"default/{base_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"lora\",\n \"max_seq_length\": 4096,\n },\n schedule={\"epochs\": 1, \"val_check_interval\": 0.1},\n batch={\n \"global_batch_size\": 64,\n \"micro_batch_size\": 1,\n \"sequence_packing\": False,\n },\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": NO_PACK_OUTPUT_NAME},\n)\n\njob_without_sequence_packing = client.customization.automodel.jobs.create(\n spec=spec, workspace=\"default\", name=JOB_NAME\n)\n\nprint(f\"Submitted job: {job_without_sequence_packing.job.name}\")\nprint(f\"Output adapter: {NO_PACK_OUTPUT_NAME}\")", "language": "python", - "source_html": "import uuid\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"my-sft-job-{job_suffix}"\n\njob_spec_no_packing = CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=1,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=4096,\n val_check_interval=0.1,\n micro_batch_size=1,\n sequence_packing=False,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n )\n)\n\njob_without_sequence_packing = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=job_spec_no_packing\n)\n\nprint(f"Job ID: {job_without_sequence_packing.name}")\nprint(f"Output model: {job_without_sequence_packing.spec.output.name}")\n" + "source_html": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"no-packing-job-{job_suffix}"\nNO_PACK_OUTPUT_NAME = f"no-packing-out-{job_suffix}"\n\nspec = AutomodelJobInput(\n model=f"default/{base_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "lora",\n "max_seq_length": 4096,\n },\n schedule={"epochs": 1, "val_check_interval": 0.1},\n batch={\n "global_batch_size": 64,\n "micro_batch_size": 1,\n "sequence_packing": False,\n },\n optimizer={"learning_rate": 5e-5},\n parallelism={"num_gpus_per_node": 1},\n output={"name": NO_PACK_OUTPUT_NAME},\n)\n\njob_without_sequence_packing = client.customization.automodel.jobs.create(\n spec=spec, workspace="default", name=JOB_NAME\n)\n\nprint(f"Submitted job: {job_without_sequence_packing.job.name}")\nprint(f"Output adapter: {NO_PACK_OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -135,9 +137,9 @@ export default { cells: [ }, { "type": "code", - "source": "# Wait for the training step to complete\njob_without_sequence_packing_status = wait_for_job(\n workspace=\"default\",\n job_name=job_without_sequence_packing.name,\n timeout=TIMEOUT_SECONDS\n)\n\nprint(f\"Validation loss: {job_without_sequence_packing_status.status_details['val_loss']:.2f}\")", + "source": "# Wait for the training step to complete\njob_without_sequence_packing_status = wait_for_job(\n workspace=\"default\",\n job_name=job_without_sequence_packing.job.name,\n timeout=TIMEOUT_SECONDS\n)\n\nno_pack_val_loss = (job_without_sequence_packing_status.status_details or {}).get(\"val_loss\")\nif no_pack_val_loss is not None:\n print(f\"Validation loss: {float(no_pack_val_loss):.2f}\")\nelse:\n print(\"Validation loss: not reported in job status\")", "language": "python", - "source_html": "# Wait for the training step to complete\njob_without_sequence_packing_status = wait_for_job(\n workspace="default",\n job_name=job_without_sequence_packing.name,\n timeout=TIMEOUT_SECONDS\n)\n\nprint(f"Validation loss: {job_without_sequence_packing_status.status_details['val_loss']:.2f}")\n" + "source_html": "# Wait for the training step to complete\njob_without_sequence_packing_status = wait_for_job(\n workspace="default",\n job_name=job_without_sequence_packing.job.name,\n timeout=TIMEOUT_SECONDS\n)\n\nno_pack_val_loss = (job_without_sequence_packing_status.status_details or {}).get("val_loss")\nif no_pack_val_loss is not None:\n print(f"Validation loss: {float(no_pack_val_loss):.2f}")\nelse:\n print("Validation loss: not reported in job status")\n" }, { "type": "markdown", @@ -146,9 +148,9 @@ export default { cells: [ }, { "type": "code", - "source": "from nemo_platform.types.jobs import PlatformJobStep\nfrom datetime import datetime\nimport pandas as pd\n\nSTEP_NAME = \"customization-training-job\"\n\ndef get_elapsed_time(step: PlatformJobStep) -> float:\n \"\"\"Calculate elapsed time in seconds from step's created_at to updated_at.\"\"\"\n created_at = datetime.fromisoformat(step.created_at.replace(\"Z\", \"+00:00\"))\n updated_at = datetime.fromisoformat(step.updated_at.replace(\"Z\", \"+00:00\"))\n return (updated_at - created_at).total_seconds()\n\nstep_with_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace=\"default\",\n job=job_with_sequence_packing.name,\n)\n\nstep_without_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace=\"default\",\n job=job_without_sequence_packing.name,\n)\n\ntime_to_complete_with_sequence_packing = get_elapsed_time(step_with_sequence_packing)\ntime_to_complete_without_sequence_packing = get_elapsed_time(step_without_sequence_packing)\n\n# Display results as a table\nresults_df = pd.DataFrame({\n \"Seq Packing Enabled\": [True, False],\n \"Val Loss\": [\n job_with_sequence_packing_status.status_details['val_loss'],\n job_without_sequence_packing_status.status_details['val_loss']\n ],\n \"Training Step Time, sec\": [\n time_to_complete_with_sequence_packing,\n time_to_complete_without_sequence_packing\n ]\n})\n\nresults_df.style.format({\"Val Loss\": \"{:.2f}\", \"Training Step Time, sec\": \"{:.0f}\"}).hide(axis='index')", + "source": "from nemo_platform.types.jobs import PlatformJobStep\nfrom datetime import datetime\nimport pandas as pd\n\nSTEP_NAME = \"training\"\n\ndef get_elapsed_time(step: PlatformJobStep) -> float:\n \"\"\"Calculate elapsed time in seconds from step's created_at to updated_at.\"\"\"\n created_at = datetime.fromisoformat(step.created_at.replace(\"Z\", \"+00:00\"))\n updated_at = datetime.fromisoformat(step.updated_at.replace(\"Z\", \"+00:00\"))\n return (updated_at - created_at).total_seconds()\n\nstep_with_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace=\"default\",\n job=job_with_sequence_packing.job.name,\n)\n\nstep_without_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace=\"default\",\n job=job_without_sequence_packing.job.name,\n)\n\ntime_to_complete_with_sequence_packing = get_elapsed_time(step_with_sequence_packing)\ntime_to_complete_without_sequence_packing = get_elapsed_time(step_without_sequence_packing)\n\n# Display results as a table\nresults_df = pd.DataFrame({\n \"Seq Packing Enabled\": [True, False],\n \"Val Loss\": [\n (job_with_sequence_packing_status.status_details or {}).get(\"val_loss\"),\n (job_without_sequence_packing_status.status_details or {}).get(\"val_loss\"),\n ],\n \"Training Step Time, sec\": [\n time_to_complete_with_sequence_packing,\n time_to_complete_without_sequence_packing\n ]\n})\n\nresults_df.style.format({\"Val Loss\": \"{:.2f}\", \"Training Step Time, sec\": \"{:.0f}\"}).hide(axis='index')", "language": "python", - "source_html": "from nemo_platform.types.jobs import PlatformJobStep\nfrom datetime import datetime\nimport pandas as pd\n\nSTEP_NAME = "customization-training-job"\n\ndef get_elapsed_time(step: PlatformJobStep) -> float:\n """Calculate elapsed time in seconds from step's created_at to updated_at."""\n created_at = datetime.fromisoformat(step.created_at.replace("Z", "+00:00"))\n updated_at = datetime.fromisoformat(step.updated_at.replace("Z", "+00:00"))\n return (updated_at - created_at).total_seconds()\n\nstep_with_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace="default",\n job=job_with_sequence_packing.name,\n)\n\nstep_without_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace="default",\n job=job_without_sequence_packing.name,\n)\n\ntime_to_complete_with_sequence_packing = get_elapsed_time(step_with_sequence_packing)\ntime_to_complete_without_sequence_packing = get_elapsed_time(step_without_sequence_packing)\n\n# Display results as a table\nresults_df = pd.DataFrame({\n "Seq Packing Enabled": [True, False],\n "Val Loss": [\n job_with_sequence_packing_status.status_details['val_loss'],\n job_without_sequence_packing_status.status_details['val_loss']\n ],\n "Training Step Time, sec": [\n time_to_complete_with_sequence_packing,\n time_to_complete_without_sequence_packing\n ]\n})\n\nresults_df.style.format({"Val Loss": "{:.2f}", "Training Step Time, sec": "{:.0f}"}).hide(axis='index')\n" + "source_html": "from nemo_platform.types.jobs import PlatformJobStep\nfrom datetime import datetime\nimport pandas as pd\n\nSTEP_NAME = "training"\n\ndef get_elapsed_time(step: PlatformJobStep) -> float:\n """Calculate elapsed time in seconds from step's created_at to updated_at."""\n created_at = datetime.fromisoformat(step.created_at.replace("Z", "+00:00"))\n updated_at = datetime.fromisoformat(step.updated_at.replace("Z", "+00:00"))\n return (updated_at - created_at).total_seconds()\n\nstep_with_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace="default",\n job=job_with_sequence_packing.job.name,\n)\n\nstep_without_sequence_packing = client.jobs.steps.retrieve(\n name=STEP_NAME,\n workspace="default",\n job=job_without_sequence_packing.job.name,\n)\n\ntime_to_complete_with_sequence_packing = get_elapsed_time(step_with_sequence_packing)\ntime_to_complete_without_sequence_packing = get_elapsed_time(step_without_sequence_packing)\n\n# Display results as a table\nresults_df = pd.DataFrame({\n "Seq Packing Enabled": [True, False],\n "Val Loss": [\n (job_with_sequence_packing_status.status_details or {}).get("val_loss"),\n (job_without_sequence_packing_status.status_details or {}).get("val_loss"),\n ],\n "Training Step Time, sec": [\n time_to_complete_with_sequence_packing,\n time_to_complete_without_sequence_packing\n ]\n})\n\nresults_df.style.format({"Val Loss": "{:.2f}", "Training Step Time, sec": "{:.0f}"}).hide(axis='index')\n" }, { "type": "markdown", diff --git a/docs/fern/components/notebooks/sft-customization-job.json b/docs/fern/components/notebooks/sft-customization-job.json index 139e4d99d1..6356affdaa 100644 --- a/docs/fern/components/notebooks/sft-customization-job.json +++ b/docs/fern/components/notebooks/sft-customization-job.json @@ -7,8 +7,8 @@ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (PyPI wrapper: pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  4. \n
\n" }, { "type": "markdown", @@ -79,9 +79,9 @@ }, { "type": "code", - "source": "# Create fileset to store SFT training data\nDATASET_NAME = \"sft-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SFT training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate training data is uploaded correctly\nprint(\"Training data:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", + "source": "# Create fileset to store SFT training data\nDATASET_NAME = \"sft-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SFT training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=f\"{DATASET_PATH}/\", # Trailing slash uploads directory contents to fileset root\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate training data is uploaded correctly\nprint(\"Training data:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", "language": "python", - "source_html": "# Create fileset to store SFT training data\nDATASET_NAME = "sft-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SFT training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate training data is uploaded correctly\nprint("Training data:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" + "source_html": "# Create fileset to store SFT training data\nDATASET_NAME = "sft-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SFT training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=f"{DATASET_PATH}/", # Trailing slash uploads directory contents to fileset root\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate training data is uploaded correctly\nprint("Training data:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" }, { "type": "markdown", @@ -107,8 +107,8 @@ }, { "type": "markdown", - "source": "### 6. Create SFT Finetuning Job\nCreate a customization job with an inline target referencing the base model and dataset filesets created in previous steps.", - "source_html": "

6. Create SFT Finetuning Job

\n

Create a customization job with an inline target referencing the base model and dataset filesets created in previous steps.

\n" + "source": "### 6. Create SFT Finetuning Job\nCreate a customization job to fine-tune all model weights using the **Automodel** backend and `AutomodelJobInput`.", + "source_html": "

6. Create SFT Finetuning Job

\n

Create a customization job to fine-tune all model weights using the Automodel backend and AutomodelJobInput.

\n" }, { "type": "markdown", @@ -117,9 +117,9 @@ }, { "type": "code", - "source": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=2,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nprint(f\"Job ID: {job.name}\")\nprint(f\"Output model: {job.spec.output.name}\")", + "source": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\nOUTPUT_NAME = f\"sft-model-{job_suffix}\"\n\nspec = AutomodelJobInput(\n model=f\"default/{base_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"all_weights\",\n \"max_seq_length\": 2048,\n },\n schedule={\"epochs\": 2},\n batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\n \"num_gpus_per_node\": 1,\n \"num_nodes\": 1,\n \"tensor_parallel_size\": 1,\n \"pipeline_parallel_size\": 1,\n },\n output={\"name\": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace=\"default\", name=JOB_NAME\n)\n\nprint(f\"Submitted job: {job.job.name}\")\nprint(f\"Output model: {OUTPUT_NAME}\")", "language": "python", - "source_html": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f"my-sft-job-{job_suffix}"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=2,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nprint(f"Job ID: {job.name}")\nprint(f"Output model: {job.spec.output.name}")\n" + "source_html": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f"my-sft-job-{job_suffix}"\nOUTPUT_NAME = f"sft-model-{job_suffix}"\n\nspec = AutomodelJobInput(\n model=f"default/{base_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "all_weights",\n "max_seq_length": 2048,\n },\n schedule={"epochs": 2},\n batch={"global_batch_size": 64, "micro_batch_size": 1},\n optimizer={"learning_rate": 5e-5},\n parallelism={\n "num_gpus_per_node": 1,\n "num_nodes": 1,\n "tensor_parallel_size": 1,\n "pipeline_parallel_size": 1,\n },\n output={"name": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace="default", name=JOB_NAME\n)\n\nprint(f"Submitted job: {job.job.name}")\nprint(f"Output model: {OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -128,9 +128,9 @@ }, { "type": "code", - "source": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace=\"default\"\n )\n\n clear_output(wait=True)\n print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get(\"step\")\n max_steps = task_details.get(\"max_steps\")\n training_phase = task_details.get(\"phase\")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n if training_phase:\n print(f\"Training Phase: {training_phase}\")\n else:\n print(\"Training step not started yet or progress info not available\")\n\n # Exit loop when job is completed (or failed/cancelled)\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n break\n\n time.sleep(10)", + "source": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.jobs.get_status(\n name=job.job.name,\n workspace=\"default\"\n )\n\n clear_output(wait=True)\n print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == \"training\":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get(\"step\")\n max_steps = task_details.get(\"max_steps\")\n training_phase = task_details.get(\"phase\")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n if training_phase:\n print(f\"Training Phase: {training_phase}\")\n else:\n print(\"Training step not started yet or progress info not available\")\n\n # Exit loop when job reaches a terminal status\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n break\n\n time.sleep(10)\n\nif status.status != \"completed\":\n raise RuntimeError(f\"Training job finished with status: {status.status}\")", "language": "python", - "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace="default"\n )\n\n clear_output(wait=True)\n print(f"Job Status: {status.model_dump_json(indent=2)}")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get("step")\n max_steps = task_details.get("max_steps")\n training_phase = task_details.get("phase")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)")\n if training_phase:\n print(f"Training Phase: {training_phase}")\n else:\n print("Training step not started yet or progress info not available")\n\n # Exit loop when job is completed (or failed/cancelled)\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n break\n\n time.sleep(10)\n" + "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.jobs.get_status(\n name=job.job.name,\n workspace="default"\n )\n\n clear_output(wait=True)\n print(f"Job Status: {status.model_dump_json(indent=2)}")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == "training":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get("step")\n max_steps = task_details.get("max_steps")\n training_phase = task_details.get("phase")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)")\n if training_phase:\n print(f"Training Phase: {training_phase}")\n else:\n print("Training step not started yet or progress info not available")\n\n # Exit loop when job reaches a terminal status\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n break\n\n time.sleep(10)\n\nif status.status != "completed":\n raise RuntimeError(f"Training job finished with status: {status.status}")\n" }, { "type": "markdown", @@ -144,25 +144,20 @@ }, { "type": "code", - "source": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))", + "source": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=OUTPUT_NAME)\nprint(model_entity.model_dump_json(indent=2))", "language": "python", - "source_html": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))\n" + "source_html": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=OUTPUT_NAME)\nprint(model_entity.model_dump_json(indent=2))\n" }, { "type": "code", - "source": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"sft-model-deployment-cfg-{deploy_suffix}\"\nDEPLOYMENT_NAME = f\"sft-model-deployment-{deploy_suffix}\"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n image_tag=\"1.15.5\",\n gpu=1,\n model_name=job.spec.output.name, # ModelEntity name from training,\n model_namespace=\"default\", # Workspace where ModelEntity lives\n additional_envs={\"NIM_MODEL_PROFILE\": \"vllm\"}\n ),\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace=\"default\"\n)\n\nprint(f\"Deployment name: {deployment.name}\")\nprint(f\"Deployment status: {deployment_status.status}\")", + "source": "# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"sft-model-deployment-cfg-{deploy_suffix}\"\nDEPLOYMENT_NAME = f\"sft-model-deployment-{deploy_suffix}\"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": OUTPUT_NAME,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n },\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace=\"default\"\n)\n\nprint(f\"Deployment name: {deployment.name}\")\nprint(f\"Deployment status: {deployment_status.status}\")", "language": "python", - "source_html": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"sft-model-deployment-cfg-{deploy_suffix}"\nDEPLOYMENT_NAME = f"sft-model-deployment-{deploy_suffix}"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name="nvcr.io/nim/nvidia/llm-nim",\n image_tag="1.15.5",\n gpu=1,\n model_name=job.spec.output.name, # ModelEntity name from training,\n model_namespace="default", # Workspace where ModelEntity lives\n additional_envs={"NIM_MODEL_PROFILE": "vllm"}\n ),\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace="default"\n)\n\nprint(f"Deployment name: {deployment.name}")\nprint(f"Deployment status: {deployment_status.status}")\n" + "source_html": "# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"sft-model-deployment-cfg-{deploy_suffix}"\nDEPLOYMENT_NAME = f"sft-model-deployment-{deploy_suffix}"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n engine="vllm",\n model_spec={\n "model_namespace": "default",\n "model_name": OUTPUT_NAME,\n },\n executor_config={\n "gpu": 1,\n "image_name": "vllm/vllm-openai",\n "image_tag": "v0.22.1",\n },\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace="default"\n)\n\nprint(f"Deployment name: {deployment.name}")\nprint(f"Deployment status: {deployment_status.status}")\n" }, { "type": "markdown", - "source": "The deployment service automatically:\n- Downloads model weights from the Files service\n- Provisions storage (PVC) for the weights\n- Configures and starts the NIM container\n\n**Multi-GPU Deployment:**\n\nFor larger models requiring multiple GPUs, configure parallelism with environment variables:\n\n```python\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=\"sft-model-config-multigpu\",\n \n nim_deployment={\n \"image_name\": \"nvcr.io/nim/nvidia/llm-nim\",\n \"image_tag\": \"1.13.1\",\n \"gpu\": 2, # Total GPUs\n \"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"2\", # Tensor parallelism\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"1\" # Pipeline parallelism\n }\n }\n)\n```", - "source_html": "

The deployment service automatically:

\n
    \n
  • Downloads model weights from the Files service
  • \n
  • Provisions storage (PVC) for the weights
  • \n
  • Configures and starts the NIM container
  • \n
\n

Multi-GPU Deployment:

\n

For larger models requiring multiple GPUs, configure parallelism with environment variables:

\n
deployment_config = client.inference.deployment_configs.create(\n    workspace="default",\n    name="sft-model-config-multigpu",\n    \n    nim_deployment={\n        "image_name": "nvcr.io/nim/nvidia/llm-nim",\n        "image_tag": "1.13.1",\n        "gpu": 2,  # Total GPUs\n        "additional_envs": {\n            "NIM_TENSOR_PARALLEL_SIZE": "2",  # Tensor parallelism\n            "NIM_PIPELINE_PARALLEL_SIZE": "1"  # Pipeline parallelism\n        }\n    }\n)\n
\n" - }, - { - "type": "markdown", - "source": "**Single-Node Constraint:** Model deployments are limited to a single node. The maximum `gpu` value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.\n\n---\n\n#### GPU Parallelism\n\nBy default, NIM uses all GPUs for tensor parallelism (TP). You can customize this behavior using the `NIM_TENSOR_PARALLEL_SIZE` and `NIM_PIPELINE_PARALLEL_SIZE` environment variables.\n\n| Strategy | Description | Best For |\n|----------|-------------|----------|\n| **Tensor Parallel (TP)** | Splits model layers across GPUs | Lowest latency |\n| **Pipeline Parallel (PP)** | Splits model depth across GPUs | Highest throughput |\n\n**Formula:** `gpu` = `NIM_TENSOR_PARALLEL_SIZE` × `NIM_PIPELINE_PARALLEL_SIZE`\n\n---\n\n#### Example Configurations\n\n**Default (TP=8, PP=1) — Lowest Latency**\n```\n\"gpu\": 8\n# NIM automatically sets NIM_TENSOR_PARALLEL_SIZE=8\n```\n\n**Balanced (TP=4, PP=2)**\n```\n\"gpu\": 8,\n\"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"4\",\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"2\"\n}\n```\n\n**Throughput Optimized (TP=2, PP=4)**\n```\n\"gpu\": 8,\n\"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"2\",\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"4\"\n}\n```", - "source_html": "

Single-Node Constraint: Model deployments are limited to a single node. The maximum gpu value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.

\n
\n

GPU Parallelism

\n

By default, NIM uses all GPUs for tensor parallelism (TP). You can customize this behavior using the NIM_TENSOR_PARALLEL_SIZE and NIM_PIPELINE_PARALLEL_SIZE environment variables.

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
StrategyDescriptionBest For
Tensor Parallel (TP)Splits model layers across GPUsLowest latency
Pipeline Parallel (PP)Splits model depth across GPUsHighest throughput
\n

Formula: gpu = NIM_TENSOR_PARALLEL_SIZE × NIM_PIPELINE_PARALLEL_SIZE

\n
\n

Example Configurations

\n

Default (TP=8, PP=1) — Lowest Latency

\n
"gpu": 8\n# NIM automatically sets NIM_TENSOR_PARALLEL_SIZE=8\n
\n

Balanced (TP=4, PP=2)

\n
"gpu": 8,\n"additional_envs": {\n    "NIM_TENSOR_PARALLEL_SIZE": "4",\n    "NIM_PIPELINE_PARALLEL_SIZE": "2"\n}\n
\n

Throughput Optimized (TP=2, PP=4)

\n
"gpu": 8,\n"additional_envs": {\n    "NIM_TENSOR_PARALLEL_SIZE": "2",\n    "NIM_PIPELINE_PARALLEL_SIZE": "4"\n}\n
\n" + "source": "The deployment service automatically:\n- Downloads model weights from the Files service\n- Provisions storage (PVC) for the weights\n- Configures and starts the vLLM container\n\n**Multi-GPU Deployment:**\n\nFor larger models requiring multiple GPUs, increase `gpu` in `executor_config`. vLLM computes tensor parallelism from the GPU count and model architecture:\n\n```python\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=\"sft-model-config-multigpu\",\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": OUTPUT_NAME,\n },\n executor_config={\n \"gpu\": 2,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n },\n)\n```\n\n**Single-Node Constraint:** Model deployments are limited to a single node. The maximum `gpu` value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.", + "source_html": "

The deployment service automatically:

\n
    \n
  • Downloads model weights from the Files service
  • \n
  • Provisions storage (PVC) for the weights
  • \n
  • Configures and starts the vLLM container
  • \n
\n

Multi-GPU Deployment:

\n

For larger models requiring multiple GPUs, increase gpu in executor_config. vLLM computes tensor parallelism from the GPU count and model architecture:

\n
deployment_config = client.inference.deployment_configs.create(\n    workspace="default",\n    name="sft-model-config-multigpu",\n    engine="vllm",\n    model_spec={\n        "model_namespace": "default",\n        "model_name": OUTPUT_NAME,\n    },\n    executor_config={\n        "gpu": 2,\n        "image_name": "vllm/vllm-openai",\n        "image_tag": "v0.22.1",\n    },\n)\n
\n

Single-Node Constraint: Model deployments are limited to a single node. The maximum gpu value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.

\n" }, { "type": "markdown", @@ -182,14 +177,14 @@ }, { "type": "code", - "source": "# Wait for deployment to be ready, then test\n# Test the fine-tuned model with a question answering prompt\ncontext = \"The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit.\"\nquestion = \"Who was the first person to walk on the Moon?\"\n\nmessages = [\n {\"role\": \"user\", \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\"}\n]\n\nresponse = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": f\"default/{job.spec.output.name}\",\n \"messages\": messages,\n \"temperature\": 0,\n \"max_tokens\": 128\n }\n)\n\nprint(\"=\" * 60)\nprint(\"MODEL EVALUATION\")\nprint(\"=\" * 60)\nprint(f\"Question: {question}\")\nprint(f\"Expected: Neil Armstrong\")\nprint(f\"Model output: {response['choices'][0]['message']['content']}\")", + "source": "# Wait for deployment to be ready, then test\n# Test the fine-tuned model with a question answering prompt\ncontext = \"The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit.\"\nquestion = \"Who was the first person to walk on the Moon?\"\n\nmessages = [\n {\"role\": \"user\", \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\"}\n]\n\nresponse = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": f\"default/{OUTPUT_NAME}\",\n \"messages\": messages,\n \"temperature\": 0,\n \"max_tokens\": 128\n }\n)\n\nprint(\"=\" * 60)\nprint(\"MODEL EVALUATION\")\nprint(\"=\" * 60)\nprint(f\"Question: {question}\")\nprint(f\"Expected: Neil Armstrong\")\nprint(f\"Model output: {response['choices'][0]['message']['content']}\")", "language": "python", - "source_html": "# Wait for deployment to be ready, then test\n# Test the fine-tuned model with a question answering prompt\ncontext = "The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit."\nquestion = "Who was the first person to walk on the Moon?"\n\nmessages = [\n {"role": "user", "content": f"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}"}\n]\n\nresponse = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment.name,\n workspace="default",\n body={\n "model": f"default/{job.spec.output.name}",\n "messages": messages,\n "temperature": 0,\n "max_tokens": 128\n }\n)\n\nprint("=" * 60)\nprint("MODEL EVALUATION")\nprint("=" * 60)\nprint(f"Question: {question}")\nprint(f"Expected: Neil Armstrong")\nprint(f"Model output: {response['choices'][0]['message']['content']}")\n" + "source_html": "# Wait for deployment to be ready, then test\n# Test the fine-tuned model with a question answering prompt\ncontext = "The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit."\nquestion = "Who was the first person to walk on the Moon?"\n\nmessages = [\n {"role": "user", "content": f"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}"}\n]\n\nresponse = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment.name,\n workspace="default",\n body={\n "model": f"default/{OUTPUT_NAME}",\n "messages": messages,\n "temperature": 0,\n "max_tokens": 128\n }\n)\n\nprint("=" * 60)\nprint("MODEL EVALUATION")\nprint("=" * 60)\nprint(f"Question: {question}")\nprint(f"Expected: Neil Armstrong")\nprint(f"Model output: {response['choices'][0]['message']['content']}")\n" }, { "type": "markdown", - "source": "#### Evaluation Best Practices\n\n**Manual Evaluation** (Recommended)\n- Test with real-world examples from your use case\n- Compare responses to base model and expected outputs\n- Verify the model exhibits desired behavior changes\n- Check edge cases and error handling\n\n**What to look for:**\n- ✅ Model follows your desired output format\n- ✅ Applies domain knowledge correctly\n- ✅ Maintains general language capabilities\n- ✅ Avoids unwanted behaviors or biases\n- ❌ Doesn't hallucinate facts not in training data\n- ❌ Doesn't produce repetitive or nonsensical outputs\n\n---\n\n## Hyperparameters\n\nFor detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n\n---\n\n\n## Troubleshooting\n\n**Job fails during model download:**\n- Verify authentication secrets are configured (refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n- Check the `model_uri` format is correct (`fileset://`)\n- Ensure you have accepted the model's terms of service on HuggingFace\n- Check job status and logs: `client.customization.jobs.retrieve(name=job.name, workspace=\"default\")`\n\n**Job fails with OOM (Out of Memory) error:**\n1. **First try:** Reduce `micro_batch_size` from 2 to 1\n2. **Still OOM:** Reduce `batch_size` from 4 to 2\n3. **Still OOM:** Reduce `max_seq_length` from 2048 to 1024 or 512\n4. **Last resort:** Increase GPU count and use `tensor_parallel_size` for model sharding\n\n**Loss curves not decreasing (underfitting):**\n- Increase training duration: `epochs: 5-10` instead of 3\n- Adjust learning rate: Try `1e-5` to `1e-4`\n- Add warmup: Set `warmup_steps` to ~10% of total training steps\n- Check data quality: Verify formatting, remove duplicates, ensure diversity\n\n**Training loss decreases but validation loss increases (overfitting):**\n- Reduce epochs: Try `epochs: 1-2` instead of 5+\n- Lower learning rate: Use `2e-5` or `1e-5`\n- Increase dataset size and diversity\n- Verify train/validation split has no data leakage\n\n**Model output quality is poor despite good training metrics:**\n- Training metrics optimize for loss, not your actual task—evaluate on real use cases\n- Review data quality, format, and diversity—metrics can be misleading with poor data\n- Try a different base model size or architecture\n- Adjust learning rate and batch size\n- Compare to baseline: Test base model to ensure fine-tuning improved performance\n\n**Deployment fails:**\n- Verify output model exists: `client.models.retrieve(name=job.spec.output.name, workspace=\"default\")`\n- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n- Ensure sufficient GPU resources available for model size\n- Verify NIM image tag `1.13.1` is compatible with your model\n\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning", - "source_html": "

Evaluation Best Practices

\n

Manual Evaluation (Recommended)

\n
    \n
  • Test with real-world examples from your use case
  • \n
  • Compare responses to base model and expected outputs
  • \n
  • Verify the model exhibits desired behavior changes
  • \n
  • Check edge cases and error handling
  • \n
\n

What to look for:

\n
    \n
  • ✅ Model follows your desired output format
  • \n
  • ✅ Applies domain knowledge correctly
  • \n
  • ✅ Maintains general language capabilities
  • \n
  • ✅ Avoids unwanted behaviors or biases
  • \n
  • ❌ Doesn't hallucinate facts not in training data
  • \n
  • ❌ Doesn't produce repetitive or nonsensical outputs
  • \n
\n
\n

Hyperparameters

\n

For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the Hyperparameter Reference.

\n
\n

Troubleshooting

\n

Job fails during model download:

\n
    \n
  • Verify authentication secrets are configured (refer to Managing Secrets)
  • \n
  • For gated HuggingFace models (Llama, Gemma), accept the license on the model page
  • \n
  • Check the model_uri format is correct (fileset://)
  • \n
  • Ensure you have accepted the model's terms of service on HuggingFace
  • \n
  • Check job status and logs: client.customization.jobs.retrieve(name=job.name, workspace="default")
  • \n
\n

Job fails with OOM (Out of Memory) error:

\n
    \n
  1. First try: Reduce micro_batch_size from 2 to 1
  2. \n
  3. Still OOM: Reduce batch_size from 4 to 2
  4. \n
  5. Still OOM: Reduce max_seq_length from 2048 to 1024 or 512
  6. \n
  7. Last resort: Increase GPU count and use tensor_parallel_size for model sharding
  8. \n
\n

Loss curves not decreasing (underfitting):

\n
    \n
  • Increase training duration: epochs: 5-10 instead of 3
  • \n
  • Adjust learning rate: Try 1e-5 to 1e-4
  • \n
  • Add warmup: Set warmup_steps to ~10% of total training steps
  • \n
  • Check data quality: Verify formatting, remove duplicates, ensure diversity
  • \n
\n

Training loss decreases but validation loss increases (overfitting):

\n
    \n
  • Reduce epochs: Try epochs: 1-2 instead of 5+
  • \n
  • Lower learning rate: Use 2e-5 or 1e-5
  • \n
  • Increase dataset size and diversity
  • \n
  • Verify train/validation split has no data leakage
  • \n
\n

Model output quality is poor despite good training metrics:

\n
    \n
  • Training metrics optimize for loss, not your actual task—evaluate on real use cases
  • \n
  • Review data quality, format, and diversity—metrics can be misleading with poor data
  • \n
  • Try a different base model size or architecture
  • \n
  • Adjust learning rate and batch size
  • \n
  • Compare to baseline: Test base model to ensure fine-tuning improved performance
  • \n
\n

Deployment fails:

\n
    \n
  • Verify output model exists: client.models.retrieve(name=job.spec.output.name, workspace="default")
  • \n
  • Check deployment logs: client.inference.deployments.get_logs(name=deployment.name, workspace="default")
  • \n
  • Ensure sufficient GPU resources available for model size
  • \n
  • Verify NIM image tag 1.13.1 is compatible with your model
  • \n
\n

Next Steps

\n\n" + "source": "#### Evaluation Best Practices\n\n**Manual Evaluation** (Recommended)\n- Test with real-world examples from your use case\n- Compare responses to base model and expected outputs\n- Verify the model exhibits desired behavior changes\n- Check edge cases and error handling\n\n**What to look for:**\n- ✅ Model follows your desired output format\n- ✅ Applies domain knowledge correctly\n- ✅ Maintains general language capabilities\n- ✅ Avoids unwanted behaviors or biases\n- ❌ Doesn't hallucinate facts not in training data\n- ❌ Doesn't produce repetitive or nonsensical outputs\n\n---\n\n## Hyperparameters\n\nFor detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n\n---\n\n\n## Troubleshooting\n\n**Job fails during model download:**\n- Verify authentication secrets are configured (refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n- For gated HuggingFace models (Llama, Gemma), accept the license on the model page (for example, [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct))\n- Confirm the model fileset uses `token_secret=hf_secret.name` for gated models\n- Check `AutomodelJobInput` references use the `workspace/name` format: `model=f\"default/{MODEL_NAME}\"` and `dataset={\"training\": f\"default/{DATASET_NAME}\"}` (for example, `default/llama-3-2-1b-base`, `default/sft-dataset`)\n- Verify the model entity points at the fileset: `fileset=f\"default/{MODEL_NAME}\"`\n- Check job status: `client.jobs.get_status(name=job.job.name, workspace=\"default\")`\n\n**Job fails with OOM (Out of Memory) error:**\n1. **First try:** Reduce `global_batch_size` from 64 to 32 or 16 in `batch={...}`\n2. **Still OOM:** Keep `micro_batch_size` at 1 (already the minimum in this tutorial)\n3. **Still OOM:** Reduce `max_seq_length` from 2048 to 1024 or 512 in `training={...}`\n4. **Last resort:** Increase `num_gpus_per_node` and `tensor_parallel_size` in `parallelism={...}`\n\n**Loss curves not decreasing (underfitting):**\n- Increase training duration: raise `epochs` from 2 to 3-5 in `schedule={...}`\n- Adjust learning rate: try `1e-4` or `1e-5` instead of the default `5e-5` in `optimizer={...}`\n- Check data quality: Verify formatting, remove duplicates, ensure diversity\n\n**Training loss decreases but validation loss increases (overfitting):**\n- Reduce `epochs` from 2 to 1 in `schedule={...}`\n- Lower `learning_rate` from `5e-5` to `2e-5` or `1e-5` in `optimizer={...}`\n- Increase dataset size and diversity\n- Verify train/validation split has no data leakage\n\n**Model output quality is poor despite good training metrics:**\n- Training metrics optimize for loss, not your actual task—evaluate on real use cases\n- Review data quality, format, and diversity—metrics can be misleading with poor data\n- Try a different base model size or architecture\n- Adjust `learning_rate` and `global_batch_size`\n- Compare to baseline: Test base model to ensure fine-tuning improved performance\n\n**Deployment fails:**\n- Verify output model exists: `client.models.retrieve(name=OUTPUT_NAME, workspace=\"default\")`\n- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n- Ensure sufficient GPU resources for `executor_config={\"gpu\": 1, ...}`\n- Verify the deployment config matches this tutorial: `engine=\"vllm\"` with `vllm/vllm-openai:v0.22.1`\n\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning", + "source_html": "

Evaluation Best Practices

\n

Manual Evaluation (Recommended)

\n
    \n
  • Test with real-world examples from your use case
  • \n
  • Compare responses to base model and expected outputs
  • \n
  • Verify the model exhibits desired behavior changes
  • \n
  • Check edge cases and error handling
  • \n
\n

What to look for:

\n
    \n
  • ✅ Model follows your desired output format
  • \n
  • ✅ Applies domain knowledge correctly
  • \n
  • ✅ Maintains general language capabilities
  • \n
  • ✅ Avoids unwanted behaviors or biases
  • \n
  • ❌ Doesn't hallucinate facts not in training data
  • \n
  • ❌ Doesn't produce repetitive or nonsensical outputs
  • \n
\n
\n

Hyperparameters

\n

For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the Hyperparameter Reference.

\n
\n

Troubleshooting

\n

Job fails during model download:

\n
    \n
  • Verify authentication secrets are configured (refer to Managing Secrets)
  • \n
  • For gated HuggingFace models (Llama, Gemma), accept the license on the model page (for example, meta-llama/Llama-3.2-1B-Instruct)
  • \n
  • Confirm the model fileset uses token_secret=hf_secret.name for gated models
  • \n
  • Check AutomodelJobInput references use the workspace/name format: model=f"default/{MODEL_NAME}" and dataset={"training": f"default/{DATASET_NAME}"} (for example, default/llama-3-2-1b-base, default/sft-dataset)
  • \n
  • Verify the model entity points at the fileset: fileset=f"default/{MODEL_NAME}"
  • \n
  • Check job status: client.jobs.get_status(name=job.job.name, workspace="default")
  • \n
\n

Job fails with OOM (Out of Memory) error:

\n
    \n
  1. First try: Reduce global_batch_size from 64 to 32 or 16 in batch={...}
  2. \n
  3. Still OOM: Keep micro_batch_size at 1 (already the minimum in this tutorial)
  4. \n
  5. Still OOM: Reduce max_seq_length from 2048 to 1024 or 512 in training={...}
  6. \n
  7. Last resort: Increase num_gpus_per_node and tensor_parallel_size in parallelism={...}
  8. \n
\n

Loss curves not decreasing (underfitting):

\n
    \n
  • Increase training duration: raise epochs from 2 to 3-5 in schedule={...}
  • \n
  • Adjust learning rate: try 1e-4 or 1e-5 instead of the default 5e-5 in optimizer={...}
  • \n
  • Check data quality: Verify formatting, remove duplicates, ensure diversity
  • \n
\n

Training loss decreases but validation loss increases (overfitting):

\n
    \n
  • Reduce epochs from 2 to 1 in schedule={...}
  • \n
  • Lower learning_rate from 5e-5 to 2e-5 or 1e-5 in optimizer={...}
  • \n
  • Increase dataset size and diversity
  • \n
  • Verify train/validation split has no data leakage
  • \n
\n

Model output quality is poor despite good training metrics:

\n
    \n
  • Training metrics optimize for loss, not your actual task—evaluate on real use cases
  • \n
  • Review data quality, format, and diversity—metrics can be misleading with poor data
  • \n
  • Try a different base model size or architecture
  • \n
  • Adjust learning_rate and global_batch_size
  • \n
  • Compare to baseline: Test base model to ensure fine-tuning improved performance
  • \n
\n

Deployment fails:

\n
    \n
  • Verify output model exists: client.models.retrieve(name=OUTPUT_NAME, workspace="default")
  • \n
  • Check deployment logs: client.inference.deployments.get_logs(name=deployment.name, workspace="default")
  • \n
  • Ensure sufficient GPU resources for executor_config={"gpu": 1, ...}
  • \n
  • Verify the deployment config matches this tutorial: engine="vllm" with vllm/vllm-openai:v0.22.1
  • \n
\n

Next Steps

\n\n" } ] } \ No newline at end of file diff --git a/docs/fern/components/notebooks/sft-customization-job.ts b/docs/fern/components/notebooks/sft-customization-job.ts index 04876dce11..850b96b06a 100644 --- a/docs/fern/components/notebooks/sft-customization-job.ts +++ b/docs/fern/components/notebooks/sft-customization-job.ts @@ -1,7 +1,9 @@ -// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -// SPDX-License-Identifier: Apache-2.0 - -/** Auto-generated by ipynb-to-fern-json.py - do not edit */ +/** + * SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + * + * Auto-generated by ipynb-to-fern-json.py - do not edit manually. + */ export default { cells: [ { "type": "markdown", @@ -10,8 +12,8 @@ export default { cells: [ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (included with `pip install nemo-platform`)", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (included with pip install nemo-platform)
  4. \n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Completed the [Quickstart](../../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n2. **Installed the Python SDK** (PyPI wrapper: `pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Completed the Quickstart to install and deploy NeMo Platform locally
  2. \n
  3. Installed the Python SDK (PyPI wrapper: pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  4. \n
\n" }, { "type": "markdown", @@ -82,9 +84,9 @@ export default { cells: [ }, { "type": "code", - "source": "# Create fileset to store SFT training data\nDATASET_NAME = \"sft-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SFT training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate training data is uploaded correctly\nprint(\"Training data:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", + "source": "# Create fileset to store SFT training data\nDATASET_NAME = \"sft-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=\"default\",\n name=DATASET_NAME,\n description=\"SFT training data\"\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=f\"{DATASET_PATH}/\", # Trailing slash uploads directory contents to fileset root\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=\"default\"\n)\n\n# Validate training data is uploaded correctly\nprint(\"Training data:\")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=\"default\").data], indent=2))", "language": "python", - "source_html": "# Create fileset to store SFT training data\nDATASET_NAME = "sft-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SFT training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=DATASET_PATH, # Local directory with your JSONL files\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate training data is uploaded correctly\nprint("Training data:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" + "source_html": "# Create fileset to store SFT training data\nDATASET_NAME = "sft-dataset"\n\ntry:\n client.files.filesets.create(\n workspace="default",\n name=DATASET_NAME,\n description="SFT training data"\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\n# Upload training data files individually to ensure correct structure\nclient.files.upload(\n local_path=f"{DATASET_PATH}/", # Trailing slash uploads directory contents to fileset root\n remote_path="",\n fileset=DATASET_NAME,\n workspace="default"\n)\n\n# Validate training data is uploaded correctly\nprint("Training data:")\nprint(json.dumps([f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace="default").data], indent=2))\n" }, { "type": "markdown", @@ -110,8 +112,8 @@ export default { cells: [ }, { "type": "markdown", - "source": "### 6. Create SFT Finetuning Job\nCreate a customization job with an inline target referencing the base model and dataset filesets created in previous steps.", - "source_html": "

6. Create SFT Finetuning Job

\n

Create a customization job with an inline target referencing the base model and dataset filesets created in previous steps.

\n" + "source": "### 6. Create SFT Finetuning Job\nCreate a customization job to fine-tune all model weights using the **Automodel** backend and `AutomodelJobInput`.", + "source_html": "

6. Create SFT Finetuning Job

\n

Create a customization job to fine-tune all model weights using the Automodel backend and AutomodelJobInput.

\n" }, { "type": "markdown", @@ -120,9 +122,9 @@ export default { cells: [ }, { "type": "code", - "source": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace=\"default\",\n spec=CustomizationJobInputParam(\n model=f\"default/{base_model.name}\",\n dataset=f\"fileset://default/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=2,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nprint(f\"Job ID: {job.name}\")\nprint(f\"Output model: {job.spec.output.name}\")", + "source": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f\"my-sft-job-{job_suffix}\"\nOUTPUT_NAME = f\"sft-model-{job_suffix}\"\n\nspec = AutomodelJobInput(\n model=f\"default/{base_model.name}\",\n dataset={\"training\": f\"default/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"all_weights\",\n \"max_seq_length\": 2048,\n },\n schedule={\"epochs\": 2},\n batch={\"global_batch_size\": 64, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": 5e-5},\n parallelism={\n \"num_gpus_per_node\": 1,\n \"num_nodes\": 1,\n \"tensor_parallel_size\": 1,\n \"pipeline_parallel_size\": 1,\n },\n output={\"name\": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace=\"default\", name=JOB_NAME\n)\n\nprint(f\"Submitted job: {job.job.name}\")\nprint(f\"Output model: {OUTPUT_NAME}\")", "language": "python", - "source_html": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f"my-sft-job-{job_suffix}"\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n workspace="default",\n spec=CustomizationJobInputParam(\n model=f"default/{base_model.name}",\n dataset=f"fileset://default/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=2,\n batch_size=64,\n learning_rate=0.00005,\n max_seq_length=2048,\n micro_batch_size=1,\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n )\n)\n\nprint(f"Job ID: {job.name}")\nprint(f"Output model: {job.spec.output.name}")\n" + "source_html": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\n\nJOB_NAME = f"my-sft-job-{job_suffix}"\nOUTPUT_NAME = f"sft-model-{job_suffix}"\n\nspec = AutomodelJobInput(\n model=f"default/{base_model.name}",\n dataset={"training": f"default/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "all_weights",\n "max_seq_length": 2048,\n },\n schedule={"epochs": 2},\n batch={"global_batch_size": 64, "micro_batch_size": 1},\n optimizer={"learning_rate": 5e-5},\n parallelism={\n "num_gpus_per_node": 1,\n "num_nodes": 1,\n "tensor_parallel_size": 1,\n "pipeline_parallel_size": 1,\n },\n output={"name": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace="default", name=JOB_NAME\n)\n\nprint(f"Submitted job: {job.job.name}")\nprint(f"Output model: {OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -131,9 +133,9 @@ export default { cells: [ }, { "type": "code", - "source": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace=\"default\"\n )\n\n clear_output(wait=True)\n print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == \"customization-training-job\":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get(\"step\")\n max_steps = task_details.get(\"max_steps\")\n training_phase = task_details.get(\"phase\")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n if training_phase:\n print(f\"Training Phase: {training_phase}\")\n else:\n print(\"Training step not started yet or progress info not available\")\n\n # Exit loop when job is completed (or failed/cancelled)\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n break\n\n time.sleep(10)", + "source": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.jobs.get_status(\n name=job.job.name,\n workspace=\"default\"\n )\n\n clear_output(wait=True)\n print(f\"Job Status: {status.model_dump_json(indent=2)}\")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == \"training\":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get(\"step\")\n max_steps = task_details.get(\"max_steps\")\n training_phase = task_details.get(\"phase\")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f\"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)\")\n if training_phase:\n print(f\"Training Phase: {training_phase}\")\n else:\n print(\"Training step not started yet or progress info not available\")\n\n # Exit loop when job reaches a terminal status\n if status.status in (\"completed\", \"failed\", \"cancelled\", \"error\"):\n print(f\"\\nJob finished with status: {status.status}\")\n break\n\n time.sleep(10)\n\nif status.status != \"completed\":\n raise RuntimeError(f\"Training job finished with status: {status.status}\")", "language": "python", - "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.customization.jobs.get_status(\n name=job.name,\n workspace="default"\n )\n\n clear_output(wait=True)\n print(f"Job Status: {status.model_dump_json(indent=2)}")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == "customization-training-job":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get("step")\n max_steps = task_details.get("max_steps")\n training_phase = task_details.get("phase")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)")\n if training_phase:\n print(f"Training Phase: {training_phase}")\n else:\n print("Training step not started yet or progress info not available")\n\n # Exit loop when job is completed (or failed/cancelled)\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n break\n\n time.sleep(10)\n" + "source_html": "import time\nfrom IPython.display import clear_output\n\n# Poll job status every 10 seconds until completed\nwhile True:\n status = client.jobs.get_status(\n name=job.job.name,\n workspace="default"\n )\n\n clear_output(wait=True)\n print(f"Job Status: {status.model_dump_json(indent=2)}")\n\n # Extract training progress from nested steps structure\n step: int | None = None\n max_steps: int | None = None\n training_phase: str | None = None\n\n for job_step in status.steps or []:\n if job_step.name == "training":\n for task in job_step.tasks or []:\n task_details = task.status_details or {}\n step = task_details.get("step")\n max_steps = task_details.get("max_steps")\n training_phase = task_details.get("phase")\n break\n break\n\n if step is not None and max_steps is not None:\n progress_pct = (step / max_steps) * 100\n print(f"Training Progress: Step {step}/{max_steps} ({progress_pct:.1f}%)")\n if training_phase:\n print(f"Training Phase: {training_phase}")\n else:\n print("Training step not started yet or progress info not available")\n\n # Exit loop when job reaches a terminal status\n if status.status in ("completed", "failed", "cancelled", "error"):\n print(f"\\nJob finished with status: {status.status}")\n break\n\n time.sleep(10)\n\nif status.status != "completed":\n raise RuntimeError(f"Training job finished with status: {status.status}")\n" }, { "type": "markdown", @@ -147,25 +149,20 @@ export default { cells: [ }, { "type": "code", - "source": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))", + "source": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=OUTPUT_NAME)\nprint(model_entity.model_dump_json(indent=2))", "language": "python", - "source_html": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=job.spec.output.name)\nprint(model_entity.model_dump_json(indent=2))\n" + "source_html": "# Validate model entity exists\nmodel_entity = client.models.retrieve(workspace='default', name=OUTPUT_NAME)\nprint(model_entity.model_dump_json(indent=2))\n" }, { "type": "code", - "source": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"sft-model-deployment-cfg-{deploy_suffix}\"\nDEPLOYMENT_NAME = f\"sft-model-deployment-{deploy_suffix}\"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name=\"nvcr.io/nim/nvidia/llm-nim\",\n image_tag=\"1.15.5\",\n gpu=1,\n model_name=job.spec.output.name, # ModelEntity name from training,\n model_namespace=\"default\", # Workspace where ModelEntity lives\n additional_envs={\"NIM_MODEL_PROFILE\": \"vllm\"}\n ),\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace=\"default\"\n)\n\nprint(f\"Deployment name: {deployment.name}\")\nprint(f\"Deployment status: {deployment_status.status}\")", + "source": "# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"sft-model-deployment-cfg-{deploy_suffix}\"\nDEPLOYMENT_NAME = f\"sft-model-deployment-{deploy_suffix}\"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=DEPLOYMENT_CONFIG_NAME,\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": OUTPUT_NAME,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n },\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace=\"default\",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace=\"default\"\n)\n\nprint(f\"Deployment name: {deployment.name}\")\nprint(f\"Deployment status: {deployment_status.status}\")", "language": "python", - "source_html": "from nemo_platform.types.inference import NIMDeploymentParam\n\n# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"sft-model-deployment-cfg-{deploy_suffix}"\nDEPLOYMENT_NAME = f"sft-model-deployment-{deploy_suffix}"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n nim_deployment=NIMDeploymentParam(\n image_name="nvcr.io/nim/nvidia/llm-nim",\n image_tag="1.15.5",\n gpu=1,\n model_name=job.spec.output.name, # ModelEntity name from training,\n model_namespace="default", # Workspace where ModelEntity lives\n additional_envs={"NIM_MODEL_PROFILE": "vllm"}\n ),\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace="default"\n)\n\nprint(f"Deployment name: {deployment.name}")\nprint(f"Deployment status: {deployment_status.status}")\n" + "source_html": "# Create deployment config\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"sft-model-deployment-cfg-{deploy_suffix}"\nDEPLOYMENT_NAME = f"sft-model-deployment-{deploy_suffix}"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace="default",\n name=DEPLOYMENT_CONFIG_NAME,\n engine="vllm",\n model_spec={\n "model_namespace": "default",\n "model_name": OUTPUT_NAME,\n },\n executor_config={\n "gpu": 1,\n "image_name": "vllm/vllm-openai",\n "image_tag": "v0.22.1",\n },\n)\n\n# Deploy model using deployment_config created above\ndeployment = client.inference.deployments.create(\n workspace="default",\n name=DEPLOYMENT_NAME,\n config=deployment_config.name\n)\n\n\n# Check deployment status\ndeployment_status = client.inference.deployments.retrieve(\n name=deployment.name,\n workspace="default"\n)\n\nprint(f"Deployment name: {deployment.name}")\nprint(f"Deployment status: {deployment_status.status}")\n" }, { "type": "markdown", - "source": "The deployment service automatically:\n- Downloads model weights from the Files service\n- Provisions storage (PVC) for the weights\n- Configures and starts the NIM container\n\n**Multi-GPU Deployment:**\n\nFor larger models requiring multiple GPUs, configure parallelism with environment variables:\n\n```python\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=\"sft-model-config-multigpu\",\n \n nim_deployment={\n \"image_name\": \"nvcr.io/nim/nvidia/llm-nim\",\n \"image_tag\": \"1.13.1\",\n \"gpu\": 2, # Total GPUs\n \"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"2\", # Tensor parallelism\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"1\" # Pipeline parallelism\n }\n }\n)\n```", - "source_html": "

The deployment service automatically:

\n
    \n
  • Downloads model weights from the Files service
  • \n
  • Provisions storage (PVC) for the weights
  • \n
  • Configures and starts the NIM container
  • \n
\n

Multi-GPU Deployment:

\n

For larger models requiring multiple GPUs, configure parallelism with environment variables:

\n
deployment_config = client.inference.deployment_configs.create(\n    workspace="default",\n    name="sft-model-config-multigpu",\n    \n    nim_deployment={\n        "image_name": "nvcr.io/nim/nvidia/llm-nim",\n        "image_tag": "1.13.1",\n        "gpu": 2,  # Total GPUs\n        "additional_envs": {\n            "NIM_TENSOR_PARALLEL_SIZE": "2",  # Tensor parallelism\n            "NIM_PIPELINE_PARALLEL_SIZE": "1"  # Pipeline parallelism\n        }\n    }\n)\n
\n" - }, - { - "type": "markdown", - "source": "**Single-Node Constraint:** Model deployments are limited to a single node. The maximum `gpu` value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.\n\n---\n\n#### GPU Parallelism\n\nBy default, NIM uses all GPUs for tensor parallelism (TP). You can customize this behavior using the `NIM_TENSOR_PARALLEL_SIZE` and `NIM_PIPELINE_PARALLEL_SIZE` environment variables.\n\n| Strategy | Description | Best For |\n|----------|-------------|----------|\n| **Tensor Parallel (TP)** | Splits model layers across GPUs | Lowest latency |\n| **Pipeline Parallel (PP)** | Splits model depth across GPUs | Highest throughput |\n\n**Formula:** `gpu` = `NIM_TENSOR_PARALLEL_SIZE` × `NIM_PIPELINE_PARALLEL_SIZE`\n\n---\n\n#### Example Configurations\n\n**Default (TP=8, PP=1) — Lowest Latency**\n```\n\"gpu\": 8\n# NIM automatically sets NIM_TENSOR_PARALLEL_SIZE=8\n```\n\n**Balanced (TP=4, PP=2)**\n```\n\"gpu\": 8,\n\"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"4\",\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"2\"\n}\n```\n\n**Throughput Optimized (TP=2, PP=4)**\n```\n\"gpu\": 8,\n\"additional_envs\": {\n \"NIM_TENSOR_PARALLEL_SIZE\": \"2\",\n \"NIM_PIPELINE_PARALLEL_SIZE\": \"4\"\n}\n```", - "source_html": "

Single-Node Constraint: Model deployments are limited to a single node. The maximum gpu value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.

\n
\n

GPU Parallelism

\n

By default, NIM uses all GPUs for tensor parallelism (TP). You can customize this behavior using the NIM_TENSOR_PARALLEL_SIZE and NIM_PIPELINE_PARALLEL_SIZE environment variables.

\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n
StrategyDescriptionBest For
Tensor Parallel (TP)Splits model layers across GPUsLowest latency
Pipeline Parallel (PP)Splits model depth across GPUsHighest throughput
\n

Formula: gpu = NIM_TENSOR_PARALLEL_SIZE × NIM_PIPELINE_PARALLEL_SIZE

\n
\n

Example Configurations

\n

Default (TP=8, PP=1) — Lowest Latency

\n
"gpu": 8\n# NIM automatically sets NIM_TENSOR_PARALLEL_SIZE=8\n
\n

Balanced (TP=4, PP=2)

\n
"gpu": 8,\n"additional_envs": {\n    "NIM_TENSOR_PARALLEL_SIZE": "4",\n    "NIM_PIPELINE_PARALLEL_SIZE": "2"\n}\n
\n

Throughput Optimized (TP=2, PP=4)

\n
"gpu": 8,\n"additional_envs": {\n    "NIM_TENSOR_PARALLEL_SIZE": "2",\n    "NIM_PIPELINE_PARALLEL_SIZE": "4"\n}\n
\n" + "source": "The deployment service automatically:\n- Downloads model weights from the Files service\n- Provisions storage (PVC) for the weights\n- Configures and starts the vLLM container\n\n**Multi-GPU Deployment:**\n\nFor larger models requiring multiple GPUs, increase `gpu` in `executor_config`. vLLM computes tensor parallelism from the GPU count and model architecture:\n\n```python\ndeployment_config = client.inference.deployment_configs.create(\n workspace=\"default\",\n name=\"sft-model-config-multigpu\",\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": \"default\",\n \"model_name\": OUTPUT_NAME,\n },\n executor_config={\n \"gpu\": 2,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n },\n)\n```\n\n**Single-Node Constraint:** Model deployments are limited to a single node. The maximum `gpu` value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.", + "source_html": "

The deployment service automatically:

\n
    \n
  • Downloads model weights from the Files service
  • \n
  • Provisions storage (PVC) for the weights
  • \n
  • Configures and starts the vLLM container
  • \n
\n

Multi-GPU Deployment:

\n

For larger models requiring multiple GPUs, increase gpu in executor_config. vLLM computes tensor parallelism from the GPU count and model architecture:

\n
deployment_config = client.inference.deployment_configs.create(\n    workspace="default",\n    name="sft-model-config-multigpu",\n    engine="vllm",\n    model_spec={\n        "model_namespace": "default",\n        "model_name": OUTPUT_NAME,\n    },\n    executor_config={\n        "gpu": 2,\n        "image_name": "vllm/vllm-openai",\n        "image_tag": "v0.22.1",\n    },\n)\n
\n

Single-Node Constraint: Model deployments are limited to a single node. The maximum gpu value depends on the total GPUs available on a single node in your cluster. Multi-node deployments are not supported.

\n" }, { "type": "markdown", @@ -185,13 +182,13 @@ export default { cells: [ }, { "type": "code", - "source": "# Wait for deployment to be ready, then test\n# Test the fine-tuned model with a question answering prompt\ncontext = \"The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit.\"\nquestion = \"Who was the first person to walk on the Moon?\"\n\nmessages = [\n {\"role\": \"user\", \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\"}\n]\n\nresponse = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": f\"default/{job.spec.output.name}\",\n \"messages\": messages,\n \"temperature\": 0,\n \"max_tokens\": 128\n }\n)\n\nprint(\"=\" * 60)\nprint(\"MODEL EVALUATION\")\nprint(\"=\" * 60)\nprint(f\"Question: {question}\")\nprint(f\"Expected: Neil Armstrong\")\nprint(f\"Model output: {response['choices'][0]['message']['content']}\")", + "source": "# Wait for deployment to be ready, then test\n# Test the fine-tuned model with a question answering prompt\ncontext = \"The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit.\"\nquestion = \"Who was the first person to walk on the Moon?\"\n\nmessages = [\n {\"role\": \"user\", \"content\": f\"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}\"}\n]\n\nresponse = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment.name,\n workspace=\"default\",\n body={\n \"model\": f\"default/{OUTPUT_NAME}\",\n \"messages\": messages,\n \"temperature\": 0,\n \"max_tokens\": 128\n }\n)\n\nprint(\"=\" * 60)\nprint(\"MODEL EVALUATION\")\nprint(\"=\" * 60)\nprint(f\"Question: {question}\")\nprint(f\"Expected: Neil Armstrong\")\nprint(f\"Model output: {response['choices'][0]['message']['content']}\")", "language": "python", - "source_html": "# Wait for deployment to be ready, then test\n# Test the fine-tuned model with a question answering prompt\ncontext = "The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit."\nquestion = "Who was the first person to walk on the Moon?"\n\nmessages = [\n {"role": "user", "content": f"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}"}\n]\n\nresponse = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment.name,\n workspace="default",\n body={\n "model": f"default/{job.spec.output.name}",\n "messages": messages,\n "temperature": 0,\n "max_tokens": 128\n }\n)\n\nprint("=" * 60)\nprint("MODEL EVALUATION")\nprint("=" * 60)\nprint(f"Question: {question}")\nprint(f"Expected: Neil Armstrong")\nprint(f"Model output: {response['choices'][0]['message']['content']}")\n" + "source_html": "# Wait for deployment to be ready, then test\n# Test the fine-tuned model with a question answering prompt\ncontext = "The Apollo 11 mission was the first manned mission to land on the Moon. It was launched on July 16, 1969, and Neil Armstrong became the first person to walk on the lunar surface on July 20, 1969. Buzz Aldrin joined him shortly after, while Michael Collins remained in lunar orbit."\nquestion = "Who was the first person to walk on the Moon?"\n\nmessages = [\n {"role": "user", "content": f"Based on the following context, answer the question.\\n\\nContext: {context}\\n\\nQuestion: {question}"}\n]\n\nresponse = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment.name,\n workspace="default",\n body={\n "model": f"default/{OUTPUT_NAME}",\n "messages": messages,\n "temperature": 0,\n "max_tokens": 128\n }\n)\n\nprint("=" * 60)\nprint("MODEL EVALUATION")\nprint("=" * 60)\nprint(f"Question: {question}")\nprint(f"Expected: Neil Armstrong")\nprint(f"Model output: {response['choices'][0]['message']['content']}")\n" }, { "type": "markdown", - "source": "#### Evaluation Best Practices\n\n**Manual Evaluation** (Recommended)\n- Test with real-world examples from your use case\n- Compare responses to base model and expected outputs\n- Verify the model exhibits desired behavior changes\n- Check edge cases and error handling\n\n**What to look for:**\n- ✅ Model follows your desired output format\n- ✅ Applies domain knowledge correctly\n- ✅ Maintains general language capabilities\n- ✅ Avoids unwanted behaviors or biases\n- ❌ Doesn't hallucinate facts not in training data\n- ❌ Doesn't produce repetitive or nonsensical outputs\n\n---\n\n## Hyperparameters\n\nFor detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n\n---\n\n\n## Troubleshooting\n\n**Job fails during model download:**\n- Verify authentication secrets are configured (refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n- For gated HuggingFace models (Llama, Gemma), accept the license on the model page\n- Check the `model_uri` format is correct (`fileset://`)\n- Ensure you have accepted the model's terms of service on HuggingFace\n- Check job status and logs: `client.customization.jobs.retrieve(name=job.name, workspace=\"default\")`\n\n**Job fails with OOM (Out of Memory) error:**\n1. **First try:** Reduce `micro_batch_size` from 2 to 1\n2. **Still OOM:** Reduce `batch_size` from 4 to 2\n3. **Still OOM:** Reduce `max_seq_length` from 2048 to 1024 or 512\n4. **Last resort:** Increase GPU count and use `tensor_parallel_size` for model sharding\n\n**Loss curves not decreasing (underfitting):**\n- Increase training duration: `epochs: 5-10` instead of 3\n- Adjust learning rate: Try `1e-5` to `1e-4`\n- Add warmup: Set `warmup_steps` to ~10% of total training steps\n- Check data quality: Verify formatting, remove duplicates, ensure diversity\n\n**Training loss decreases but validation loss increases (overfitting):**\n- Reduce epochs: Try `epochs: 1-2` instead of 5+\n- Lower learning rate: Use `2e-5` or `1e-5`\n- Increase dataset size and diversity\n- Verify train/validation split has no data leakage\n\n**Model output quality is poor despite good training metrics:**\n- Training metrics optimize for loss, not your actual task—evaluate on real use cases\n- Review data quality, format, and diversity—metrics can be misleading with poor data\n- Try a different base model size or architecture\n- Adjust learning rate and batch size\n- Compare to baseline: Test base model to ensure fine-tuning improved performance\n\n**Deployment fails:**\n- Verify output model exists: `client.models.retrieve(name=job.spec.output.name, workspace=\"default\")`\n- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n- Ensure sufficient GPU resources available for model size\n- Verify NIM image tag `1.13.1` is compatible with your model\n\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning", - "source_html": "

Evaluation Best Practices

\n

Manual Evaluation (Recommended)

\n
    \n
  • Test with real-world examples from your use case
  • \n
  • Compare responses to base model and expected outputs
  • \n
  • Verify the model exhibits desired behavior changes
  • \n
  • Check edge cases and error handling
  • \n
\n

What to look for:

\n
    \n
  • ✅ Model follows your desired output format
  • \n
  • ✅ Applies domain knowledge correctly
  • \n
  • ✅ Maintains general language capabilities
  • \n
  • ✅ Avoids unwanted behaviors or biases
  • \n
  • ❌ Doesn't hallucinate facts not in training data
  • \n
  • ❌ Doesn't produce repetitive or nonsensical outputs
  • \n
\n
\n

Hyperparameters

\n

For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the Hyperparameter Reference.

\n
\n

Troubleshooting

\n

Job fails during model download:

\n
    \n
  • Verify authentication secrets are configured (refer to Managing Secrets)
  • \n
  • For gated HuggingFace models (Llama, Gemma), accept the license on the model page
  • \n
  • Check the model_uri format is correct (fileset://)
  • \n
  • Ensure you have accepted the model's terms of service on HuggingFace
  • \n
  • Check job status and logs: client.customization.jobs.retrieve(name=job.name, workspace="default")
  • \n
\n

Job fails with OOM (Out of Memory) error:

\n
    \n
  1. First try: Reduce micro_batch_size from 2 to 1
  2. \n
  3. Still OOM: Reduce batch_size from 4 to 2
  4. \n
  5. Still OOM: Reduce max_seq_length from 2048 to 1024 or 512
  6. \n
  7. Last resort: Increase GPU count and use tensor_parallel_size for model sharding
  8. \n
\n

Loss curves not decreasing (underfitting):

\n
    \n
  • Increase training duration: epochs: 5-10 instead of 3
  • \n
  • Adjust learning rate: Try 1e-5 to 1e-4
  • \n
  • Add warmup: Set warmup_steps to ~10% of total training steps
  • \n
  • Check data quality: Verify formatting, remove duplicates, ensure diversity
  • \n
\n

Training loss decreases but validation loss increases (overfitting):

\n
    \n
  • Reduce epochs: Try epochs: 1-2 instead of 5+
  • \n
  • Lower learning rate: Use 2e-5 or 1e-5
  • \n
  • Increase dataset size and diversity
  • \n
  • Verify train/validation split has no data leakage
  • \n
\n

Model output quality is poor despite good training metrics:

\n
    \n
  • Training metrics optimize for loss, not your actual task—evaluate on real use cases
  • \n
  • Review data quality, format, and diversity—metrics can be misleading with poor data
  • \n
  • Try a different base model size or architecture
  • \n
  • Adjust learning rate and batch size
  • \n
  • Compare to baseline: Test base model to ensure fine-tuning improved performance
  • \n
\n

Deployment fails:

\n
    \n
  • Verify output model exists: client.models.retrieve(name=job.spec.output.name, workspace="default")
  • \n
  • Check deployment logs: client.inference.deployments.get_logs(name=deployment.name, workspace="default")
  • \n
  • Ensure sufficient GPU resources available for model size
  • \n
  • Verify NIM image tag 1.13.1 is compatible with your model
  • \n
\n

Next Steps

\n\n" + "source": "#### Evaluation Best Practices\n\n**Manual Evaluation** (Recommended)\n- Test with real-world examples from your use case\n- Compare responses to base model and expected outputs\n- Verify the model exhibits desired behavior changes\n- Check edge cases and error handling\n\n**What to look for:**\n- ✅ Model follows your desired output format\n- ✅ Applies domain knowledge correctly\n- ✅ Maintains general language capabilities\n- ✅ Avoids unwanted behaviors or biases\n- ❌ Doesn't hallucinate facts not in training data\n- ❌ Doesn't produce repetitive or nonsensical outputs\n\n---\n\n## Hyperparameters\n\nFor detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the [Hyperparameter Reference](../manage-customization-jobs/hyperparameters.md).\n\n---\n\n\n## Troubleshooting\n\n**Job fails during model download:**\n- Verify authentication secrets are configured (refer to [Managing Secrets](../../get-started/concepts/manage-secrets.md))\n- For gated HuggingFace models (Llama, Gemma), accept the license on the model page (for example, [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct))\n- Confirm the model fileset uses `token_secret=hf_secret.name` for gated models\n- Check `AutomodelJobInput` references use the `workspace/name` format: `model=f\"default/{MODEL_NAME}\"` and `dataset={\"training\": f\"default/{DATASET_NAME}\"}` (for example, `default/llama-3-2-1b-base`, `default/sft-dataset`)\n- Verify the model entity points at the fileset: `fileset=f\"default/{MODEL_NAME}\"`\n- Check job status: `client.jobs.get_status(name=job.job.name, workspace=\"default\")`\n\n**Job fails with OOM (Out of Memory) error:**\n1. **First try:** Reduce `global_batch_size` from 64 to 32 or 16 in `batch={...}`\n2. **Still OOM:** Keep `micro_batch_size` at 1 (already the minimum in this tutorial)\n3. **Still OOM:** Reduce `max_seq_length` from 2048 to 1024 or 512 in `training={...}`\n4. **Last resort:** Increase `num_gpus_per_node` and `tensor_parallel_size` in `parallelism={...}`\n\n**Loss curves not decreasing (underfitting):**\n- Increase training duration: raise `epochs` from 2 to 3-5 in `schedule={...}`\n- Adjust learning rate: try `1e-4` or `1e-5` instead of the default `5e-5` in `optimizer={...}`\n- Check data quality: Verify formatting, remove duplicates, ensure diversity\n\n**Training loss decreases but validation loss increases (overfitting):**\n- Reduce `epochs` from 2 to 1 in `schedule={...}`\n- Lower `learning_rate` from `5e-5` to `2e-5` or `1e-5` in `optimizer={...}`\n- Increase dataset size and diversity\n- Verify train/validation split has no data leakage\n\n**Model output quality is poor despite good training metrics:**\n- Training metrics optimize for loss, not your actual task—evaluate on real use cases\n- Review data quality, format, and diversity—metrics can be misleading with poor data\n- Try a different base model size or architecture\n- Adjust `learning_rate` and `global_batch_size`\n- Compare to baseline: Test base model to ensure fine-tuning improved performance\n\n**Deployment fails:**\n- Verify output model exists: `client.models.retrieve(name=OUTPUT_NAME, workspace=\"default\")`\n- Check deployment logs: `client.inference.deployments.get_logs(name=deployment.name, workspace=\"default\")`\n- Ensure sufficient GPU resources for `executor_config={\"gpu\": 1, ...}`\n- Verify the deployment config matches this tutorial: `engine=\"vllm\"` with `vllm/vllm-openai:v0.22.1`\n\n\n## Next Steps\n\n- [Monitor training metrics](fine-tune-metrics) in detail\n- [Evaluate your fine-tuned model](../../evaluator/index) using the Evaluator service\n- Learn about [LoRA customization](./lora-customization-job) for resource-efficient fine-tuning", + "source_html": "

Evaluation Best Practices

\n

Manual Evaluation (Recommended)

\n
    \n
  • Test with real-world examples from your use case
  • \n
  • Compare responses to base model and expected outputs
  • \n
  • Verify the model exhibits desired behavior changes
  • \n
  • Check edge cases and error handling
  • \n
\n

What to look for:

\n
    \n
  • ✅ Model follows your desired output format
  • \n
  • ✅ Applies domain knowledge correctly
  • \n
  • ✅ Maintains general language capabilities
  • \n
  • ✅ Avoids unwanted behaviors or biases
  • \n
  • ❌ Doesn't hallucinate facts not in training data
  • \n
  • ❌ Doesn't produce repetitive or nonsensical outputs
  • \n
\n
\n

Hyperparameters

\n

For detailed information on all available hyperparameters, recommended values, and tuning guidance, refer to the Hyperparameter Reference.

\n
\n

Troubleshooting

\n

Job fails during model download:

\n
    \n
  • Verify authentication secrets are configured (refer to Managing Secrets)
  • \n
  • For gated HuggingFace models (Llama, Gemma), accept the license on the model page (for example, meta-llama/Llama-3.2-1B-Instruct)
  • \n
  • Confirm the model fileset uses token_secret=hf_secret.name for gated models
  • \n
  • Check AutomodelJobInput references use the workspace/name format: model=f"default/{MODEL_NAME}" and dataset={"training": f"default/{DATASET_NAME}"} (for example, default/llama-3-2-1b-base, default/sft-dataset)
  • \n
  • Verify the model entity points at the fileset: fileset=f"default/{MODEL_NAME}"
  • \n
  • Check job status: client.jobs.get_status(name=job.job.name, workspace="default")
  • \n
\n

Job fails with OOM (Out of Memory) error:

\n
    \n
  1. First try: Reduce global_batch_size from 64 to 32 or 16 in batch={...}
  2. \n
  3. Still OOM: Keep micro_batch_size at 1 (already the minimum in this tutorial)
  4. \n
  5. Still OOM: Reduce max_seq_length from 2048 to 1024 or 512 in training={...}
  6. \n
  7. Last resort: Increase num_gpus_per_node and tensor_parallel_size in parallelism={...}
  8. \n
\n

Loss curves not decreasing (underfitting):

\n
    \n
  • Increase training duration: raise epochs from 2 to 3-5 in schedule={...}
  • \n
  • Adjust learning rate: try 1e-4 or 1e-5 instead of the default 5e-5 in optimizer={...}
  • \n
  • Check data quality: Verify formatting, remove duplicates, ensure diversity
  • \n
\n

Training loss decreases but validation loss increases (overfitting):

\n
    \n
  • Reduce epochs from 2 to 1 in schedule={...}
  • \n
  • Lower learning_rate from 5e-5 to 2e-5 or 1e-5 in optimizer={...}
  • \n
  • Increase dataset size and diversity
  • \n
  • Verify train/validation split has no data leakage
  • \n
\n

Model output quality is poor despite good training metrics:

\n
    \n
  • Training metrics optimize for loss, not your actual task—evaluate on real use cases
  • \n
  • Review data quality, format, and diversity—metrics can be misleading with poor data
  • \n
  • Try a different base model size or architecture
  • \n
  • Adjust learning_rate and global_batch_size
  • \n
  • Compare to baseline: Test base model to ensure fine-tuning improved performance
  • \n
\n

Deployment fails:

\n
    \n
  • Verify output model exists: client.models.retrieve(name=OUTPUT_NAME, workspace="default")
  • \n
  • Check deployment logs: client.inference.deployments.get_logs(name=deployment.name, workspace="default")
  • \n
  • Ensure sufficient GPU resources for executor_config={"gpu": 1, ...}
  • \n
  • Verify the deployment config matches this tutorial: engine="vllm" with vllm/vllm-openai:v0.22.1
  • \n
\n

Next Steps

\n\n" } ] }; diff --git a/docs/fern/components/notebooks/tool-calling.json b/docs/fern/components/notebooks/tool-calling.json index d8af8b353c..0d7d76d052 100644 --- a/docs/fern/components/notebooks/tool-calling.json +++ b/docs/fern/components/notebooks/tool-calling.json @@ -7,8 +7,8 @@ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Installed the Python SDK with the Data Designer extra** (`uv pip install nemo-platform[data-designer]`)\n1. **Completed the [Quickstart](../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n1. (Optional if running outside of Quickstart) **Authenticated with the platform** using the CLI:\n ```bash\n nemo auth login\n ```\n For non-default URLs: `nemo auth login --base-url `\n1. Set your **Hugging Face token** as an environment variable before running the notebook (`export HF_TOKEN=\"\"`)\n1. **Accepted the dataset license** for [Salesforce/xlam-function-calling-60k](https://huggingface.co/datasets/Salesforce/xlam-function-calling-60k) on Hugging Face (used for evaluation) ", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Installed the Python SDK with the Data Designer extra (uv pip install nemo-platform[data-designer])
  2. \n
  3. Completed the Quickstart to install and deploy NeMo Platform locally
  4. \n
  5. (Optional if running outside of Quickstart) Authenticated with the platform using the CLI:
    nemo auth login\n
    \nFor non-default URLs: nemo auth login --base-url <YOUR_NMP_BASE_URL>
  6. \n
  7. Set your Hugging Face token as an environment variable before running the notebook (export HF_TOKEN="<your-token>")
  8. \n
  9. Accepted the dataset license for Salesforce/xlam-function-calling-60k on Hugging Face (used for evaluation)
  10. \n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Installed the Python SDK with Data Designer support** (PyPI wrapper: `uv pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n1. **Completed the [Quickstart](../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n1. (Optional if running outside of Quickstart) **Authenticated with the platform** using the CLI:\n ```bash\n nemo auth login\n ```\n For non-default URLs: `nemo auth login --base-url `\n1. Set your **Hugging Face token** as an environment variable before running the notebook (`export HF_TOKEN=\"\"`)\n1. **Accepted the dataset license** for [Salesforce/xlam-function-calling-60k](https://huggingface.co/datasets/Salesforce/xlam-function-calling-60k) on Hugging Face (used for evaluation) ", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Installed the Python SDK with Data Designer support (PyPI wrapper: uv pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  2. \n
  3. Completed the Quickstart to install and deploy NeMo Platform locally
  4. \n
  5. (Optional if running outside of Quickstart) Authenticated with the platform using the CLI:
    nemo auth login\n
    \nFor non-default URLs: nemo auth login --base-url <YOUR_NMP_BASE_URL>
  6. \n
  7. Set your Hugging Face token as an environment variable before running the notebook (export HF_TOKEN="<your-token>")
  8. \n
  9. Accepted the dataset license for Salesforce/xlam-function-calling-60k on Hugging Face (used for evaluation)
  10. \n
\n" }, { "type": "markdown", @@ -138,9 +138,9 @@ }, { "type": "code", - "source": "DATASET_NAME = \"tool-calling-dataset\"\n\ntry:\n client.files.filesets.create(\n name=DATASET_NAME,\n description=\"synthetic tool-calling training and validation data in OpenAI chat format\",\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path=\"\",\n fileset=DATASET_NAME,\n)\n\nprint(\"\\nTraining data files:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=DATASET_NAME).data],\n indent=2,\n))", + "source": "DATASET_NAME = \"tool-calling-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=WORKSPACE,\n name=DATASET_NAME,\n description=\"synthetic tool-calling training and validation data in OpenAI chat format\",\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=WORKSPACE,\n)\n\nprint(\"\\nTraining data files:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=WORKSPACE).data],\n indent=2,\n))", "language": "python", - "source_html": "DATASET_NAME = "tool-calling-dataset"\n\ntry:\n client.files.filesets.create(\n name=DATASET_NAME,\n description="synthetic tool-calling training and validation data in OpenAI chat format",\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path="",\n fileset=DATASET_NAME,\n)\n\nprint("\\nTraining data files:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=DATASET_NAME).data],\n indent=2,\n))\n" + "source_html": "DATASET_NAME = "tool-calling-dataset"\n\ntry:\n client.files.filesets.create(\n workspace=WORKSPACE,\n name=DATASET_NAME,\n description="synthetic tool-calling training and validation data in OpenAI chat format",\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path="",\n fileset=DATASET_NAME,\n workspace=WORKSPACE,\n)\n\nprint("\\nTraining data files:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=WORKSPACE).data],\n indent=2,\n))\n" }, { "type": "markdown", @@ -149,9 +149,9 @@ }, { "type": "code", - "source": "try:\n client.files.filesets.create(\n name=EVAL_DATASET_NAME,\n description=\"synthetic tool-calling evaluation data (messages, tools, ground-truth tool_calls)\",\n )\n print(f\"Created fileset: {EVAL_DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{EVAL_DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=EVAL_DATASET_PATH,\n remote_path=\"\",\n fileset=EVAL_DATASET_NAME,\n)\n\nprint(\"\\nEvaluation data files:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=EVAL_DATASET_NAME).data],\n indent=2,\n))", + "source": "try:\n client.files.filesets.create(\n workspace=WORKSPACE,\n name=EVAL_DATASET_NAME,\n description=\"synthetic tool-calling evaluation data (messages, tools, ground-truth tool_calls)\",\n )\n print(f\"Created fileset: {EVAL_DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{EVAL_DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=EVAL_DATASET_PATH,\n remote_path=\"\",\n fileset=EVAL_DATASET_NAME,\n workspace=WORKSPACE,\n)\n\nprint(\"\\nEvaluation data files:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=EVAL_DATASET_NAME, workspace=WORKSPACE).data],\n indent=2,\n))", "language": "python", - "source_html": "try:\n client.files.filesets.create(\n name=EVAL_DATASET_NAME,\n description="synthetic tool-calling evaluation data (messages, tools, ground-truth tool_calls)",\n )\n print(f"Created fileset: {EVAL_DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{EVAL_DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=EVAL_DATASET_PATH,\n remote_path="",\n fileset=EVAL_DATASET_NAME,\n)\n\nprint("\\nEvaluation data files:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=EVAL_DATASET_NAME).data],\n indent=2,\n))\n" + "source_html": "try:\n client.files.filesets.create(\n workspace=WORKSPACE,\n name=EVAL_DATASET_NAME,\n description="synthetic tool-calling evaluation data (messages, tools, ground-truth tool_calls)",\n )\n print(f"Created fileset: {EVAL_DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{EVAL_DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=EVAL_DATASET_PATH,\n remote_path="",\n fileset=EVAL_DATASET_NAME,\n workspace=WORKSPACE,\n)\n\nprint("\\nEvaluation data files:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=EVAL_DATASET_NAME, workspace=WORKSPACE).data],\n indent=2,\n))\n" }, { "type": "markdown", @@ -160,9 +160,9 @@ }, { "type": "code", - "source": "HF_TOKEN = os.environ.get(\"HF_TOKEN\")\nif HF_TOKEN is None:\n raise ValueError(\"HF_TOKEN is not set.\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n name=name,\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name)\n\n\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(hf_secret.model_dump_json(indent=2))", + "source": "HF_TOKEN = os.environ.get(\"HF_TOKEN\")\nif HF_TOKEN is None:\n raise ValueError(\"HF_TOKEN is not set.\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n workspace=WORKSPACE,\n name=name,\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name, workspace=WORKSPACE)\n\n\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(hf_secret.model_dump_json(indent=2))", "language": "python", - "source_html": "HF_TOKEN = os.environ.get("HF_TOKEN")\nif HF_TOKEN is None:\n raise ValueError("HF_TOKEN is not set.")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f"{label} is not set")\n try:\n secret = client.secrets.create(\n name=name,\n value=value,\n )\n print(f"Created secret: {name}")\n return secret\n except ConflictError:\n print(f"Secret '{name}' already exists, continuing...")\n return client.secrets.retrieve(name=name)\n\n\nhf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN")\nprint(hf_secret.model_dump_json(indent=2))\n" + "source_html": "HF_TOKEN = os.environ.get("HF_TOKEN")\nif HF_TOKEN is None:\n raise ValueError("HF_TOKEN is not set.")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f"{label} is not set")\n try:\n secret = client.secrets.create(\n workspace=WORKSPACE,\n name=name,\n value=value,\n )\n print(f"Created secret: {name}")\n return secret\n except ConflictError:\n print(f"Secret '{name}' already exists, continuing...")\n return client.secrets.retrieve(name=name, workspace=WORKSPACE)\n\n\nhf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN")\nprint(hf_secret.model_dump_json(indent=2))\n" }, { "type": "markdown", @@ -171,20 +171,20 @@ }, { "type": "code", - "source": "import time\n\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = \"meta-llama/Llama-3.2-1B-Instruct\"\nMODEL_NAME = \"llama-3-2-1b-base\"\n\ntry:\n base_model_fs = client.files.filesets.create(\n name=MODEL_NAME,\n description=\"Llama 3.2 1B Instruct base model from HuggingFace\",\n storage=HuggingfaceStorageConfigParam(\n type=\"huggingface\",\n repo_id=HF_REPO_ID,\n repo_type=\"model\",\n token_secret=hf_secret.name,\n ),\n metadata={\n \"model\": {\n \"tool_calling\": {\n \"tool_call_parser\": \"llama3_json\",\n \"auto_tool_choice\": True,\n }\n }\n },\n )\n print(f\"Created base model fileset: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model fileset already exists. Skipping creation.\")\n base_model_fs = client.files.filesets.retrieve(\n name=MODEL_NAME,\n )\n\ntry:\n base_model = client.models.create(\n name=MODEL_NAME,\n fileset=f\"{WORKSPACE}/{MODEL_NAME}\",\n )\n print(f\"Created Model Entity: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model already exists. Updating fileset if different.\")\n base_model = client.models.update(\n name=MODEL_NAME,\n fileset=f\"{WORKSPACE}/{MODEL_NAME}\",\n )\n\nprint(f\"\\nBase model fileset: fileset://{WORKSPACE}/{base_model.name}\")\nprint(\"Base model fileset files list:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=MODEL_NAME).data],\n indent=2,\n))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint(\"\\nWaiting for ModelSpec to be populated...\")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n time.sleep(2)\n base_model = client.models.retrieve(\n name=MODEL_NAME,\n )\n\nprint(f\"ModelSpec populated: {base_model.spec.model_dump()}\")", + "source": "import time\n\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = \"meta-llama/Llama-3.2-1B-Instruct\"\nMODEL_NAME = \"llama-3-2-1b-base\"\n\ntry:\n base_model_fs = client.files.filesets.create(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n description=\"Llama 3.2 1B Instruct base model from HuggingFace\",\n storage=HuggingfaceStorageConfigParam(\n type=\"huggingface\",\n repo_id=HF_REPO_ID,\n repo_type=\"model\",\n token_secret=hf_secret.name,\n ),\n metadata={\n \"model\": {\n \"tool_calling\": {\n \"tool_call_parser\": \"llama3_json\",\n \"auto_tool_choice\": True,\n }\n }\n },\n )\n print(f\"Created base model fileset: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model fileset already exists. Skipping creation.\")\n base_model_fs = client.files.filesets.retrieve(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n )\n\ntry:\n base_model = client.models.create(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n fileset=f\"{WORKSPACE}/{MODEL_NAME}\",\n )\n print(f\"Created Model Entity: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model already exists. Updating fileset if different.\")\n base_model = client.models.update(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n fileset=f\"{WORKSPACE}/{MODEL_NAME}\",\n )\n\nprint(f\"\\nBase model fileset: fileset://{WORKSPACE}/{base_model.name}\")\nprint(\"Base model fileset files list:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=WORKSPACE).data],\n indent=2,\n))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint(\"\\nWaiting for ModelSpec to be populated...\")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n time.sleep(2)\n base_model = client.models.retrieve(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n )\n\nprint(f\"ModelSpec populated: {base_model.spec.model_dump()}\")", "language": "python", - "source_html": "import time\n\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = "meta-llama/Llama-3.2-1B-Instruct"\nMODEL_NAME = "llama-3-2-1b-base"\n\ntry:\n base_model_fs = client.files.filesets.create(\n name=MODEL_NAME,\n description="Llama 3.2 1B Instruct base model from HuggingFace",\n storage=HuggingfaceStorageConfigParam(\n type="huggingface",\n repo_id=HF_REPO_ID,\n repo_type="model",\n token_secret=hf_secret.name,\n ),\n metadata={\n "model": {\n "tool_calling": {\n "tool_call_parser": "llama3_json",\n "auto_tool_choice": True,\n }\n }\n },\n )\n print(f"Created base model fileset: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model fileset already exists. Skipping creation.")\n base_model_fs = client.files.filesets.retrieve(\n name=MODEL_NAME,\n )\n\ntry:\n base_model = client.models.create(\n name=MODEL_NAME,\n fileset=f"{WORKSPACE}/{MODEL_NAME}",\n )\n print(f"Created Model Entity: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model already exists. Updating fileset if different.")\n base_model = client.models.update(\n name=MODEL_NAME,\n fileset=f"{WORKSPACE}/{MODEL_NAME}",\n )\n\nprint(f"\\nBase model fileset: fileset://{WORKSPACE}/{base_model.name}")\nprint("Base model fileset files list:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=MODEL_NAME).data],\n indent=2,\n))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint("\\nWaiting for ModelSpec to be populated...")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds")\n time.sleep(2)\n base_model = client.models.retrieve(\n name=MODEL_NAME,\n )\n\nprint(f"ModelSpec populated: {base_model.spec.model_dump()}")\n" + "source_html": "import time\n\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = "meta-llama/Llama-3.2-1B-Instruct"\nMODEL_NAME = "llama-3-2-1b-base"\n\ntry:\n base_model_fs = client.files.filesets.create(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n description="Llama 3.2 1B Instruct base model from HuggingFace",\n storage=HuggingfaceStorageConfigParam(\n type="huggingface",\n repo_id=HF_REPO_ID,\n repo_type="model",\n token_secret=hf_secret.name,\n ),\n metadata={\n "model": {\n "tool_calling": {\n "tool_call_parser": "llama3_json",\n "auto_tool_choice": True,\n }\n }\n },\n )\n print(f"Created base model fileset: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model fileset already exists. Skipping creation.")\n base_model_fs = client.files.filesets.retrieve(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n )\n\ntry:\n base_model = client.models.create(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n fileset=f"{WORKSPACE}/{MODEL_NAME}",\n )\n print(f"Created Model Entity: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model already exists. Updating fileset if different.")\n base_model = client.models.update(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n fileset=f"{WORKSPACE}/{MODEL_NAME}",\n )\n\nprint(f"\\nBase model fileset: fileset://{WORKSPACE}/{base_model.name}")\nprint("Base model fileset files list:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=WORKSPACE).data],\n indent=2,\n))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint("\\nWaiting for ModelSpec to be populated...")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds")\n time.sleep(2)\n base_model = client.models.retrieve(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n )\n\nprint(f"ModelSpec populated: {base_model.spec.model_dump()}")\n" }, { "type": "markdown", - "source": "## 4. Create LoRA Fine-Tuning Job\n\nCreate a customization job using SFT training with LoRA PEFT. The `peft=LoRaParamsParam()` parameter enables LoRA instead of full-weight fine-tuning.\n\n**LoRA defaults** (can be overridden in `LoRaParamsParam()`):\n- `rank`: LoRA rank (dimensionality of the low-rank matrices)\n- `alpha`: Scaling factor for LoRA updates\n- `target_modules`: Which model layers to apply LoRA to", - "source_html": "

4. Create LoRA Fine-Tuning Job

\n

Create a customization job using SFT training with LoRA PEFT. The peft=LoRaParamsParam() parameter enables LoRA instead of full-weight fine-tuning.

\n

LoRA defaults (can be overridden in LoRaParamsParam()):

\n
    \n
  • rank: LoRA rank (dimensionality of the low-rank matrices)
  • \n
  • alpha: Scaling factor for LoRA updates
  • \n
  • target_modules: Which model layers to apply LoRA to
  • \n
\n" + "source": "## 4. Create LoRA Fine-Tuning Job\n\nCreate an **Automodel** customization job using `AutomodelJobInput` with `finetuning_type: lora`. After training completes, deploy the base model with LoRA support in the next section.", + "source_html": "

4. Create LoRA Fine-Tuning Job

\n

Create an Automodel customization job using AutomodelJobInput with finetuning_type: lora. After training completes, deploy the base model with LoRA support in the next section.

\n" }, { "type": "code", - "source": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n LoRaParamsParam,\n DeploymentParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"tool-calling-lora-{job_suffix}\"\n\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n spec=CustomizationJobInputParam(\n model=f\"{WORKSPACE}/{base_model.name}\",\n dataset=f\"fileset://{WORKSPACE}/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=4,\n batch_size=4,\n learning_rate=0.0001,\n max_seq_length=2048,\n micro_batch_size=1,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n deployment_config=DeploymentParamsParam(\n lora_enabled=True,\n ),\n ),\n)\n\nprint(job.model_dump_json(indent=2))", + "source": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"tool-calling-lora-{job_suffix}\"\nOUTPUT_NAME = f\"tool-calling-adapter-{job_suffix}\"\n\nspec = AutomodelJobInput(\n model=f\"{WORKSPACE}/{base_model.name}\",\n dataset={\"training\": f\"{WORKSPACE}/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"lora\",\n \"max_seq_length\": 2048,\n },\n schedule={\"epochs\": 4},\n batch={\"global_batch_size\": 4, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": 1e-4},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace=WORKSPACE, name=JOB_NAME\n)\n\nprint(f\"Submitted job: {job.job.name}\")\nprint(f\"Output adapter: {OUTPUT_NAME}\")", "language": "python", - "source_html": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n LoRaParamsParam,\n DeploymentParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"tool-calling-lora-{job_suffix}"\n\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n spec=CustomizationJobInputParam(\n model=f"{WORKSPACE}/{base_model.name}",\n dataset=f"fileset://{WORKSPACE}/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=4,\n batch_size=4,\n learning_rate=0.0001,\n max_seq_length=2048,\n micro_batch_size=1,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n deployment_config=DeploymentParamsParam(\n lora_enabled=True,\n ),\n ),\n)\n\nprint(job.model_dump_json(indent=2))\n" + "source_html": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"tool-calling-lora-{job_suffix}"\nOUTPUT_NAME = f"tool-calling-adapter-{job_suffix}"\n\nspec = AutomodelJobInput(\n model=f"{WORKSPACE}/{base_model.name}",\n dataset={"training": f"{WORKSPACE}/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "lora",\n "max_seq_length": 2048,\n },\n schedule={"epochs": 4},\n batch={"global_batch_size": 4, "micro_batch_size": 1},\n optimizer={"learning_rate": 1e-4},\n parallelism={"num_gpus_per_node": 1},\n output={"name": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace=WORKSPACE, name=JOB_NAME\n)\n\nprint(f"Submitted job: {job.job.name}")\nprint(f"Output adapter: {OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -193,26 +193,26 @@ }, { "type": "code", - "source": "import time\nfrom IPython.display import clear_output\n\nTERMINAL_JOB_STATUSES = {\"completed\", \"cancelled\", \"error\"}\n\n\ndef wait_for_job(poll_fn, label, timeout_minutes=60, poll_interval=10, display_fn=None):\n \"\"\"Poll a platform job until it reaches a terminal state.\n\n Both Customizer and Evaluator use the Core Jobs service, so the same\n PlatformJobStatus values apply: created, pending, active, cancelled,\n cancelling, error, completed, paused, pausing, resuming.\n\n Args:\n poll_fn: Callable returning a status object with .name and .status attributes.\n label: Display label for progress output.\n timeout_minutes: Maximum time to wait before returning.\n poll_interval: Seconds between polls.\n display_fn: Optional callable(status) to print extra details after the header.\n\n Returns:\n The final status object.\n \"\"\"\n start = time.time()\n timeout = timeout_minutes * 60\n\n while True:\n status = poll_fn()\n elapsed = time.time() - start\n elapsed_min, elapsed_sec = divmod(int(elapsed), 60)\n\n clear_output(wait=True)\n print(f\"[{label}] Job: {status.name}\")\n print(f\"[{label}] Status: {status.status}\")\n print(f\"[{label}] Elapsed: {elapsed_min}m {elapsed_sec}s\")\n\n if display_fn:\n display_fn(status)\n\n if status.status in TERMINAL_JOB_STATUSES:\n print(f\"\\n[{label}] Job finished: {status.status}\")\n return status\n\n if elapsed > timeout:\n print(f\"\\n[{label}] Timeout after {timeout_minutes} minutes\")\n return status\n\n time.sleep(poll_interval)\n\n\ndef training_progress(status):\n \"\"\"Extract and display training step progress.\"\"\"\n for step in status.steps or []:\n if step.name == \"customization-training-job\":\n for task in step.tasks or []:\n details = task.status_details or {}\n s, mx = details.get(\"step\"), details.get(\"max_steps\")\n if s is not None and mx is not None:\n print(f\"Training: Step {s}/{mx} ({(s / mx) * 100:.1f}%)\")\n if phase := details.get(\"phase\"):\n print(f\"Phase: {phase}\")\n return\n print(\"Training step not started yet\")\n\nprint(\"Defined wait_for_job helper function\")", + "source": "import time\nfrom IPython.display import clear_output\n\nTERMINAL_JOB_STATUSES = {\"completed\", \"failed\", \"cancelled\", \"error\"}\n\n\ndef wait_for_job(poll_fn, label, timeout_minutes=60, poll_interval=10, display_fn=None):\n \"\"\"Poll a platform job until it reaches a terminal state.\n\n Both Customizer and Evaluator use the Core Jobs service, so the same\n PlatformJobStatus values apply: created, pending, active, cancelled,\n cancelling, error, completed, paused, pausing, resuming.\n\n Args:\n poll_fn: Callable returning a status object with .name and .status attributes.\n label: Display label for progress output.\n timeout_minutes: Maximum time to wait before returning.\n poll_interval: Seconds between polls.\n display_fn: Optional callable(status) to print extra details after the header.\n\n Returns:\n The final status object.\n \"\"\"\n start = time.time()\n timeout = timeout_minutes * 60\n\n while True:\n status = poll_fn()\n elapsed = time.time() - start\n elapsed_min, elapsed_sec = divmod(int(elapsed), 60)\n\n clear_output(wait=True)\n print(f\"[{label}] Job: {status.name}\")\n print(f\"[{label}] Status: {status.status}\")\n print(f\"[{label}] Elapsed: {elapsed_min}m {elapsed_sec}s\")\n\n if display_fn:\n display_fn(status)\n\n if status.status in TERMINAL_JOB_STATUSES:\n print(f\"\\n[{label}] Job finished: {status.status}\")\n return status\n\n if elapsed > timeout:\n print(f\"\\n[{label}] Timeout after {timeout_minutes} minutes\")\n return status\n\n time.sleep(poll_interval)\n\n\ndef training_progress(status):\n \"\"\"Extract and display training step progress.\"\"\"\n for step in status.steps or []:\n if step.name == \"training\":\n for task in step.tasks or []:\n details = task.status_details or {}\n s, mx = details.get(\"step\"), details.get(\"max_steps\")\n if s is not None and mx is not None:\n print(f\"Training: Step {s}/{mx} ({(s / mx) * 100:.1f}%)\")\n if phase := details.get(\"phase\"):\n print(f\"Phase: {phase}\")\n return\n print(\"Training step not started yet\")\n\nprint(\"Defined wait_for_job helper function\")", "language": "python", - "source_html": "import time\nfrom IPython.display import clear_output\n\nTERMINAL_JOB_STATUSES = {"completed", "cancelled", "error"}\n\n\ndef wait_for_job(poll_fn, label, timeout_minutes=60, poll_interval=10, display_fn=None):\n """Poll a platform job until it reaches a terminal state.\n\n Both Customizer and Evaluator use the Core Jobs service, so the same\n PlatformJobStatus values apply: created, pending, active, cancelled,\n cancelling, error, completed, paused, pausing, resuming.\n\n Args:\n poll_fn: Callable returning a status object with .name and .status attributes.\n label: Display label for progress output.\n timeout_minutes: Maximum time to wait before returning.\n poll_interval: Seconds between polls.\n display_fn: Optional callable(status) to print extra details after the header.\n\n Returns:\n The final status object.\n """\n start = time.time()\n timeout = timeout_minutes * 60\n\n while True:\n status = poll_fn()\n elapsed = time.time() - start\n elapsed_min, elapsed_sec = divmod(int(elapsed), 60)\n\n clear_output(wait=True)\n print(f"[{label}] Job: {status.name}")\n print(f"[{label}] Status: {status.status}")\n print(f"[{label}] Elapsed: {elapsed_min}m {elapsed_sec}s")\n\n if display_fn:\n display_fn(status)\n\n if status.status in TERMINAL_JOB_STATUSES:\n print(f"\\n[{label}] Job finished: {status.status}")\n return status\n\n if elapsed > timeout:\n print(f"\\n[{label}] Timeout after {timeout_minutes} minutes")\n return status\n\n time.sleep(poll_interval)\n\n\ndef training_progress(status):\n """Extract and display training step progress."""\n for step in status.steps or []:\n if step.name == "customization-training-job":\n for task in step.tasks or []:\n details = task.status_details or {}\n s, mx = details.get("step"), details.get("max_steps")\n if s is not None and mx is not None:\n print(f"Training: Step {s}/{mx} ({(s / mx) * 100:.1f}%)")\n if phase := details.get("phase"):\n print(f"Phase: {phase}")\n return\n print("Training step not started yet")\n\nprint("Defined wait_for_job helper function")\n" + "source_html": "import time\nfrom IPython.display import clear_output\n\nTERMINAL_JOB_STATUSES = {"completed", "failed", "cancelled", "error"}\n\n\ndef wait_for_job(poll_fn, label, timeout_minutes=60, poll_interval=10, display_fn=None):\n """Poll a platform job until it reaches a terminal state.\n\n Both Customizer and Evaluator use the Core Jobs service, so the same\n PlatformJobStatus values apply: created, pending, active, cancelled,\n cancelling, error, completed, paused, pausing, resuming.\n\n Args:\n poll_fn: Callable returning a status object with .name and .status attributes.\n label: Display label for progress output.\n timeout_minutes: Maximum time to wait before returning.\n poll_interval: Seconds between polls.\n display_fn: Optional callable(status) to print extra details after the header.\n\n Returns:\n The final status object.\n """\n start = time.time()\n timeout = timeout_minutes * 60\n\n while True:\n status = poll_fn()\n elapsed = time.time() - start\n elapsed_min, elapsed_sec = divmod(int(elapsed), 60)\n\n clear_output(wait=True)\n print(f"[{label}] Job: {status.name}")\n print(f"[{label}] Status: {status.status}")\n print(f"[{label}] Elapsed: {elapsed_min}m {elapsed_sec}s")\n\n if display_fn:\n display_fn(status)\n\n if status.status in TERMINAL_JOB_STATUSES:\n print(f"\\n[{label}] Job finished: {status.status}")\n return status\n\n if elapsed > timeout:\n print(f"\\n[{label}] Timeout after {timeout_minutes} minutes")\n return status\n\n time.sleep(poll_interval)\n\n\ndef training_progress(status):\n """Extract and display training step progress."""\n for step in status.steps or []:\n if step.name == "training":\n for task in step.tasks or []:\n details = task.status_details or {}\n s, mx = details.get("step"), details.get("max_steps")\n if s is not None and mx is not None:\n print(f"Training: Step {s}/{mx} ({(s / mx) * 100:.1f}%)")\n if phase := details.get("phase"):\n print(f"Phase: {phase}")\n return\n print("Training step not started yet")\n\nprint("Defined wait_for_job helper function")\n" }, { "type": "code", - "source": "job_status = wait_for_job(\n poll_fn=lambda: client.customization.jobs.get_status(name=job.name),\n label=\"Training\",\n timeout_minutes=120,\n display_fn=training_progress,\n)", + "source": "job_status = wait_for_job(\n poll_fn=lambda: client.jobs.get_status(name=job.job.name, workspace=WORKSPACE),\n label=\"Training\",\n timeout_minutes=120,\n display_fn=training_progress,\n)\n\nif job_status.status != \"completed\":\n raise RuntimeError(f\"Training job finished with status: {job_status.status}\")", "language": "python", - "source_html": "job_status = wait_for_job(\n poll_fn=lambda: client.customization.jobs.get_status(name=job.name),\n label="Training",\n timeout_minutes=120,\n display_fn=training_progress,\n)\n" + "source_html": "job_status = wait_for_job(\n poll_fn=lambda: client.jobs.get_status(name=job.job.name, workspace=WORKSPACE),\n label="Training",\n timeout_minutes=120,\n display_fn=training_progress,\n)\n\nif job_status.status != "completed":\n raise RuntimeError(f"Training job finished with status: {job_status.status}")\n" }, { "type": "markdown", - "source": "## 5. Verify Auto-Deployed Model\n\nSince we set `lora_enabled=True` in the customization job's `deployment_config`, the platform automatically creates a NIM deployment for the base model after training completes. The LoRA adapter is attached to the base model entity (enabled by default) and the deployment serves both the base weights and the adapter through a single NIM instance.", - "source_html": "

5. Verify Auto-Deployed Model

\n

Since we set lora_enabled=True in the customization job's deployment_config, the platform automatically creates a NIM deployment for the base model after training completes. The LoRA adapter is attached to the base model entity (enabled by default) and the deployment serves both the base weights and the adapter through a single NIM instance.

\n" + "source": "## 5. Deploy Fine-Tuned Model\n\nAfter training completes, verify the LoRA adapter is attached to the base model entity, then create a NIM deployment with `lora_enabled=True` so both the base weights and adapter are served through a single deployment.", + "source_html": "

5. Deploy Fine-Tuned Model

\n

After training completes, verify the LoRA adapter is attached to the base model entity, then create a NIM deployment with lora_enabled=True so both the base weights and adapter are served through a single deployment.

\n" }, { "type": "code", - "source": "ADAPTER_NAME = job.spec.output.name\nprint(f\"Looking for adapter: {ADAPTER_NAME}\")\n\n# The adapter may not be attached to the model entity immediately after\n# training completes — poll until it appears.\nADAPTER_TIMEOUT = 120\nadapter_start = time.time()\nadapter = None\nwhile time.time() - adapter_start < ADAPTER_TIMEOUT:\n base_model = client.models.retrieve(name=MODEL_NAME)\n matches = [a for a in (base_model.adapters or []) if a.name == ADAPTER_NAME]\n if matches:\n adapter = matches[0]\n break\n print(f\"Adapter not yet attached, retrying... ({int(time.time() - adapter_start)}s)\")\n time.sleep(5)\n\nif adapter is None:\n raise TimeoutError(\n f\"Adapter '{ADAPTER_NAME}' not found on model '{MODEL_NAME}' within {ADAPTER_TIMEOUT}s\"\n )\n\nprint(f\"Base model: {base_model.name}\")\nprint(f\"Adapter:\\n{adapter.model_dump_json(indent=2)}\")", + "source": "ADAPTER_NAME = OUTPUT_NAME\nprint(f\"Looking for adapter: {ADAPTER_NAME}\")\n\n# The adapter may not be attached to the model entity immediately after\n# training completes — poll until it appears.\nADAPTER_TIMEOUT = 120\nadapter_start = time.time()\nadapter = None\nwhile time.time() - adapter_start < ADAPTER_TIMEOUT:\n base_model = client.models.retrieve(name=MODEL_NAME, workspace=WORKSPACE)\n matches = [a for a in (base_model.adapters or []) if a.name == ADAPTER_NAME]\n if matches:\n adapter = matches[0]\n break\n print(f\"Adapter not yet attached, retrying... ({int(time.time() - adapter_start)}s)\")\n time.sleep(5)\n\nif adapter is None:\n raise TimeoutError(\n f\"Adapter '{ADAPTER_NAME}' not found on model '{MODEL_NAME}' within {ADAPTER_TIMEOUT}s\"\n )\n\nprint(f\"Base model: {base_model.name}\")\nprint(f\"Adapter:\\n{adapter.model_dump_json(indent=2)}\")\n\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"tool-calling-deploy-cfg-{deploy_suffix}\"\ndeployment_name = f\"tool-calling-deploy-{deploy_suffix}\"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=WORKSPACE,\n name=DEPLOYMENT_CONFIG_NAME,\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": WORKSPACE,\n \"model_name\": MODEL_NAME,\n \"lora_enabled\": True,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n \"additional_args\": [\"--max-lora-rank\", \"32\"],\n },\n)\n\ndeployment = client.inference.deployments.create(\n workspace=WORKSPACE,\n name=deployment_name,\n config=deployment_config.name,\n)\n\nprint(f\"Deployment status: {deployment.status}\")", "language": "python", - "source_html": "ADAPTER_NAME = job.spec.output.name\nprint(f"Looking for adapter: {ADAPTER_NAME}")\n\n# The adapter may not be attached to the model entity immediately after\n# training completes — poll until it appears.\nADAPTER_TIMEOUT = 120\nadapter_start = time.time()\nadapter = None\nwhile time.time() - adapter_start < ADAPTER_TIMEOUT:\n base_model = client.models.retrieve(name=MODEL_NAME)\n matches = [a for a in (base_model.adapters or []) if a.name == ADAPTER_NAME]\n if matches:\n adapter = matches[0]\n break\n print(f"Adapter not yet attached, retrying... ({int(time.time() - adapter_start)}s)")\n time.sleep(5)\n\nif adapter is None:\n raise TimeoutError(\n f"Adapter '{ADAPTER_NAME}' not found on model '{MODEL_NAME}' within {ADAPTER_TIMEOUT}s"\n )\n\nprint(f"Base model: {base_model.name}")\nprint(f"Adapter:\\n{adapter.model_dump_json(indent=2)}")\n" + "source_html": "ADAPTER_NAME = OUTPUT_NAME\nprint(f"Looking for adapter: {ADAPTER_NAME}")\n\n# The adapter may not be attached to the model entity immediately after\n# training completes — poll until it appears.\nADAPTER_TIMEOUT = 120\nadapter_start = time.time()\nadapter = None\nwhile time.time() - adapter_start < ADAPTER_TIMEOUT:\n base_model = client.models.retrieve(name=MODEL_NAME, workspace=WORKSPACE)\n matches = [a for a in (base_model.adapters or []) if a.name == ADAPTER_NAME]\n if matches:\n adapter = matches[0]\n break\n print(f"Adapter not yet attached, retrying... ({int(time.time() - adapter_start)}s)")\n time.sleep(5)\n\nif adapter is None:\n raise TimeoutError(\n f"Adapter '{ADAPTER_NAME}' not found on model '{MODEL_NAME}' within {ADAPTER_TIMEOUT}s"\n )\n\nprint(f"Base model: {base_model.name}")\nprint(f"Adapter:\\n{adapter.model_dump_json(indent=2)}")\n\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"tool-calling-deploy-cfg-{deploy_suffix}"\ndeployment_name = f"tool-calling-deploy-{deploy_suffix}"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=WORKSPACE,\n name=DEPLOYMENT_CONFIG_NAME,\n engine="vllm",\n model_spec={\n "model_namespace": WORKSPACE,\n "model_name": MODEL_NAME,\n "lora_enabled": True,\n },\n executor_config={\n "gpu": 1,\n "image_name": "vllm/vllm-openai",\n "image_tag": "v0.22.1",\n "additional_args": ["--max-lora-rank", "32"],\n },\n)\n\ndeployment = client.inference.deployments.create(\n workspace=WORKSPACE,\n name=deployment_name,\n config=deployment_config.name,\n)\n\nprint(f"Deployment status: {deployment.status}")\n" }, { "type": "markdown", @@ -221,9 +221,9 @@ }, { "type": "code", - "source": "DEPLOYMENT_NAME = f\"sft-deploy-{MODEL_NAME}\"\n\nTIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f\"Monitoring deployment '{DEPLOYMENT_NAME}'...\")\nprint(f\"Timeout: {TIMEOUT_MINUTES} minutes\\n\")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=DEPLOYMENT_NAME,\n )\n\n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n\n clear_output(wait=True)\n print(f\"Deployment: {DEPLOYMENT_NAME}\")\n print(f\"Status: {deployment_status.status}\")\n print(f\"Elapsed time: {elapsed_min}m {elapsed_sec}s\")\n\n if deployment_status.status == \"READY\":\n print(\"\\nDeployment is ready!\")\n if not client.models.wait_for_gateway(DEPLOYMENT_NAME, workspace=WORKSPACE, timeout=60):\n raise RuntimeError(\"Inference gateway did not become ready\")\n break\n\n if deployment_status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\", \"DELETED\"):\n raise RuntimeError(f\"Deployment failed with status: {deployment_status.status}\")\n\n if elapsed > timeout_seconds:\n raise TimeoutError(f\"Deployment timeout after {TIMEOUT_MINUTES} minutes\")\n\n time.sleep(15)", + "source": "TIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f\"Monitoring deployment '{deployment_name}'...\")\nprint(f\"Timeout: {TIMEOUT_MINUTES} minutes\\n\")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=deployment_name,\n workspace=WORKSPACE,\n )\n\n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n\n clear_output(wait=True)\n print(f\"Deployment: {deployment_name}\")\n print(f\"Status: {deployment_status.status}\")\n print(f\"Elapsed time: {elapsed_min}m {elapsed_sec}s\")\n\n if deployment_status.status in (\"RUNNING\", \"READY\"):\n print(\"\\nDeployment is ready!\")\n if not client.models.wait_for_gateway(deployment_name, workspace=WORKSPACE, timeout=60):\n raise RuntimeError(\"Inference gateway did not become ready\")\n break\n\n if deployment_status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\", \"DELETED\"):\n raise RuntimeError(f\"Deployment failed with status: {deployment_status.status}\")\n\n if elapsed > timeout_seconds:\n raise TimeoutError(f\"Deployment timeout after {TIMEOUT_MINUTES} minutes\")\n\n time.sleep(15)", "language": "python", - "source_html": "DEPLOYMENT_NAME = f"sft-deploy-{MODEL_NAME}"\n\nTIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f"Monitoring deployment '{DEPLOYMENT_NAME}'...")\nprint(f"Timeout: {TIMEOUT_MINUTES} minutes\\n")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=DEPLOYMENT_NAME,\n )\n\n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n\n clear_output(wait=True)\n print(f"Deployment: {DEPLOYMENT_NAME}")\n print(f"Status: {deployment_status.status}")\n print(f"Elapsed time: {elapsed_min}m {elapsed_sec}s")\n\n if deployment_status.status == "READY":\n print("\\nDeployment is ready!")\n if not client.models.wait_for_gateway(DEPLOYMENT_NAME, workspace=WORKSPACE, timeout=60):\n raise RuntimeError("Inference gateway did not become ready")\n break\n\n if deployment_status.status in ("FAILED", "ERROR", "TERMINATED", "LOST", "DELETED"):\n raise RuntimeError(f"Deployment failed with status: {deployment_status.status}")\n\n if elapsed > timeout_seconds:\n raise TimeoutError(f"Deployment timeout after {TIMEOUT_MINUTES} minutes")\n\n time.sleep(15)\n" + "source_html": "TIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f"Monitoring deployment '{deployment_name}'...")\nprint(f"Timeout: {TIMEOUT_MINUTES} minutes\\n")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=deployment_name,\n workspace=WORKSPACE,\n )\n\n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n\n clear_output(wait=True)\n print(f"Deployment: {deployment_name}")\n print(f"Status: {deployment_status.status}")\n print(f"Elapsed time: {elapsed_min}m {elapsed_sec}s")\n\n if deployment_status.status in ("RUNNING", "READY"):\n print("\\nDeployment is ready!")\n if not client.models.wait_for_gateway(deployment_name, workspace=WORKSPACE, timeout=60):\n raise RuntimeError("Inference gateway did not become ready")\n break\n\n if deployment_status.status in ("FAILED", "ERROR", "TERMINATED", "LOST", "DELETED"):\n raise RuntimeError(f"Deployment failed with status: {deployment_status.status}")\n\n if elapsed > timeout_seconds:\n raise TimeoutError(f"Deployment timeout after {TIMEOUT_MINUTES} minutes")\n\n time.sleep(15)\n" }, { "type": "markdown", @@ -232,9 +232,9 @@ }, { "type": "code", - "source": "test_messages = [\n {\"role\": \"user\", \"content\": \"Calculate the factorial of 12 using math functions.\"},\n]\n\ntest_tools = [\n {\n \"type\": \"function\",\n \"function\": {\n \"name\": \"math_factorial\",\n \"description\": \"Calculate the factorial of a given number.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"number\": {\n \"type\": \"integer\",\n \"description\": \"The number for which factorial needs to be calculated.\",\n }\n },\n \"required\": [\"number\"],\n },\n },\n }\n]\n\n\ndef test_tool_calling(model_name: str, label: str):\n \"\"\"Send a tool calling request and display the response.\"\"\"\n response = client.inference.gateway.model.post(\n \"v1/chat/completions\",\n name=model_name,\n body={\n \"messages\": test_messages,\n \"tools\": test_tools,\n \"tool_choice\": \"auto\",\n \"temperature\": 0,\n \"max_tokens\": 256,\n },\n )\n\n print(f\"{'=' * 60}\")\n print(f\" {label}\")\n print(f\"{'=' * 60}\")\n print(json.dumps(response, indent=2))\n\n\ntest_tool_calling(MODEL_NAME, \"BASE MODEL (before fine-tuning)\")\ntest_tool_calling(ADAPTER_NAME, \"FINE-TUNED MODEL (after LoRA)\")", + "source": "test_messages = [\n {\"role\": \"user\", \"content\": \"Calculate the factorial of 12 using math functions.\"},\n]\n\ntest_tools = [\n {\n \"type\": \"function\",\n \"function\": {\n \"name\": \"math_factorial\",\n \"description\": \"Calculate the factorial of a given number.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"number\": {\n \"type\": \"integer\",\n \"description\": \"The number for which factorial needs to be calculated.\",\n }\n },\n \"required\": [\"number\"],\n },\n },\n }\n]\n\n\ndef test_tool_calling(model_name: str, label: str):\n \"\"\"Send a tool calling request and display the response.\"\"\"\n response = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment_name,\n workspace=WORKSPACE,\n body={\n \"model\": model_name,\n \"messages\": test_messages,\n \"tools\": test_tools,\n \"tool_choice\": \"auto\",\n \"temperature\": 0,\n \"max_tokens\": 256,\n },\n )\n\n print(f\"{'=' * 60}\")\n print(f\" {label}\")\n print(f\"{'=' * 60}\")\n print(json.dumps(response, indent=2))\n\n\ntest_tool_calling(MODEL_NAME, \"BASE MODEL (before fine-tuning)\")\ntest_tool_calling(ADAPTER_NAME, \"FINE-TUNED MODEL (after LoRA)\")", "language": "python", - "source_html": "test_messages = [\n {"role": "user", "content": "Calculate the factorial of 12 using math functions."},\n]\n\ntest_tools = [\n {\n "type": "function",\n "function": {\n "name": "math_factorial",\n "description": "Calculate the factorial of a given number.",\n "parameters": {\n "type": "object",\n "properties": {\n "number": {\n "type": "integer",\n "description": "The number for which factorial needs to be calculated.",\n }\n },\n "required": ["number"],\n },\n },\n }\n]\n\n\ndef test_tool_calling(model_name: str, label: str):\n """Send a tool calling request and display the response."""\n response = client.inference.gateway.model.post(\n "v1/chat/completions",\n name=model_name,\n body={\n "messages": test_messages,\n "tools": test_tools,\n "tool_choice": "auto",\n "temperature": 0,\n "max_tokens": 256,\n },\n )\n\n print(f"{'=' * 60}")\n print(f" {label}")\n print(f"{'=' * 60}")\n print(json.dumps(response, indent=2))\n\n\ntest_tool_calling(MODEL_NAME, "BASE MODEL (before fine-tuning)")\ntest_tool_calling(ADAPTER_NAME, "FINE-TUNED MODEL (after LoRA)")\n" + "source_html": "test_messages = [\n {"role": "user", "content": "Calculate the factorial of 12 using math functions."},\n]\n\ntest_tools = [\n {\n "type": "function",\n "function": {\n "name": "math_factorial",\n "description": "Calculate the factorial of a given number.",\n "parameters": {\n "type": "object",\n "properties": {\n "number": {\n "type": "integer",\n "description": "The number for which factorial needs to be calculated.",\n }\n },\n "required": ["number"],\n },\n },\n }\n]\n\n\ndef test_tool_calling(model_name: str, label: str):\n """Send a tool calling request and display the response."""\n response = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment_name,\n workspace=WORKSPACE,\n body={\n "model": model_name,\n "messages": test_messages,\n "tools": test_tools,\n "tool_choice": "auto",\n "temperature": 0,\n "max_tokens": 256,\n },\n )\n\n print(f"{'=' * 60}")\n print(f" {label}")\n print(f"{'=' * 60}")\n print(json.dumps(response, indent=2))\n\n\ntest_tool_calling(MODEL_NAME, "BASE MODEL (before fine-tuning)")\ntest_tool_calling(ADAPTER_NAME, "FINE-TUNED MODEL (after LoRA)")\n" }, { "type": "markdown", diff --git a/docs/fern/components/notebooks/tool-calling.ts b/docs/fern/components/notebooks/tool-calling.ts index 65d63f5e36..70fb87d896 100644 --- a/docs/fern/components/notebooks/tool-calling.ts +++ b/docs/fern/components/notebooks/tool-calling.ts @@ -1,7 +1,9 @@ -// SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -// SPDX-License-Identifier: Apache-2.0 - -/** Auto-generated by ipynb-to-fern-json.py - do not edit */ +/** + * SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + * + * Auto-generated by ipynb-to-fern-json.py - do not edit manually. + */ export default { cells: [ { "type": "markdown", @@ -10,8 +12,8 @@ export default { cells: [ }, { "type": "markdown", - "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Installed the Python SDK with the Data Designer extra** (`uv pip install nemo-platform[data-designer]`)\n1. **Completed the [Quickstart](../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n1. (Optional if running outside of Quickstart) **Authenticated with the platform** using the CLI:\n ```bash\n nemo auth login\n ```\n For non-default URLs: `nemo auth login --base-url `\n1. Set your **Hugging Face token** as an environment variable before running the notebook (`export HF_TOKEN=\"\"`)\n1. **Accepted the dataset license** for [Salesforce/xlam-function-calling-60k](https://huggingface.co/datasets/Salesforce/xlam-function-calling-60k) on Hugging Face (used for evaluation) ", - "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Installed the Python SDK with the Data Designer extra (uv pip install nemo-platform[data-designer])
  2. \n
  3. Completed the Quickstart to install and deploy NeMo Platform locally
  4. \n
  5. (Optional if running outside of Quickstart) Authenticated with the platform using the CLI:
    nemo auth login\n
    \nFor non-default URLs: nemo auth login --base-url <YOUR_NMP_BASE_URL>
  6. \n
  7. Set your Hugging Face token as an environment variable before running the notebook (export HF_TOKEN="<your-token>")
  8. \n
  9. Accepted the dataset license for Salesforce/xlam-function-calling-60k on Hugging Face (used for evaluation)
  10. \n
\n" + "source": "## Prerequisites\n\nBefore starting this tutorial, ensure you have:\n\n1. **Installed the Python SDK with Data Designer support** (PyPI wrapper: `uv pip install \"nemo-platform[all]\"`; source checkout: run `make bootstrap` from the repository root)\n1. **Completed the [Quickstart](../get-started/quickstart.md)** to install and deploy NeMo Platform locally\n1. (Optional if running outside of Quickstart) **Authenticated with the platform** using the CLI:\n ```bash\n nemo auth login\n ```\n For non-default URLs: `nemo auth login --base-url `\n1. Set your **Hugging Face token** as an environment variable before running the notebook (`export HF_TOKEN=\"\"`)\n1. **Accepted the dataset license** for [Salesforce/xlam-function-calling-60k](https://huggingface.co/datasets/Salesforce/xlam-function-calling-60k) on Hugging Face (used for evaluation) ", + "source_html": "

Prerequisites

\n

Before starting this tutorial, ensure you have:

\n
    \n
  1. Installed the Python SDK with Data Designer support (PyPI wrapper: uv pip install "nemo-platform[all]"; source checkout: run make bootstrap from the repository root)
  2. \n
  3. Completed the Quickstart to install and deploy NeMo Platform locally
  4. \n
  5. (Optional if running outside of Quickstart) Authenticated with the platform using the CLI:
    nemo auth login\n
    \nFor non-default URLs: nemo auth login --base-url <YOUR_NMP_BASE_URL>
  6. \n
  7. Set your Hugging Face token as an environment variable before running the notebook (export HF_TOKEN="<your-token>")
  8. \n
  9. Accepted the dataset license for Salesforce/xlam-function-calling-60k on Hugging Face (used for evaluation)
  10. \n
\n" }, { "type": "markdown", @@ -141,9 +143,9 @@ export default { cells: [ }, { "type": "code", - "source": "DATASET_NAME = \"tool-calling-dataset\"\n\ntry:\n client.files.filesets.create(\n name=DATASET_NAME,\n description=\"synthetic tool-calling training and validation data in OpenAI chat format\",\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path=\"\",\n fileset=DATASET_NAME,\n)\n\nprint(\"\\nTraining data files:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=DATASET_NAME).data],\n indent=2,\n))", + "source": "DATASET_NAME = \"tool-calling-dataset\"\n\ntry:\n client.files.filesets.create(\n workspace=WORKSPACE,\n name=DATASET_NAME,\n description=\"synthetic tool-calling training and validation data in OpenAI chat format\",\n )\n print(f\"Created fileset: {DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path=\"\",\n fileset=DATASET_NAME,\n workspace=WORKSPACE,\n)\n\nprint(\"\\nTraining data files:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=WORKSPACE).data],\n indent=2,\n))", "language": "python", - "source_html": "DATASET_NAME = "tool-calling-dataset"\n\ntry:\n client.files.filesets.create(\n name=DATASET_NAME,\n description="synthetic tool-calling training and validation data in OpenAI chat format",\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path="",\n fileset=DATASET_NAME,\n)\n\nprint("\\nTraining data files:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=DATASET_NAME).data],\n indent=2,\n))\n" + "source_html": "DATASET_NAME = "tool-calling-dataset"\n\ntry:\n client.files.filesets.create(\n workspace=WORKSPACE,\n name=DATASET_NAME,\n description="synthetic tool-calling training and validation data in OpenAI chat format",\n )\n print(f"Created fileset: {DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=DATASET_PATH,\n remote_path="",\n fileset=DATASET_NAME,\n workspace=WORKSPACE,\n)\n\nprint("\\nTraining data files:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=DATASET_NAME, workspace=WORKSPACE).data],\n indent=2,\n))\n" }, { "type": "markdown", @@ -152,9 +154,9 @@ export default { cells: [ }, { "type": "code", - "source": "try:\n client.files.filesets.create(\n name=EVAL_DATASET_NAME,\n description=\"synthetic tool-calling evaluation data (messages, tools, ground-truth tool_calls)\",\n )\n print(f\"Created fileset: {EVAL_DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{EVAL_DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=EVAL_DATASET_PATH,\n remote_path=\"\",\n fileset=EVAL_DATASET_NAME,\n)\n\nprint(\"\\nEvaluation data files:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=EVAL_DATASET_NAME).data],\n indent=2,\n))", + "source": "try:\n client.files.filesets.create(\n workspace=WORKSPACE,\n name=EVAL_DATASET_NAME,\n description=\"synthetic tool-calling evaluation data (messages, tools, ground-truth tool_calls)\",\n )\n print(f\"Created fileset: {EVAL_DATASET_NAME}\")\nexcept ConflictError:\n print(f\"Fileset '{EVAL_DATASET_NAME}' already exists, continuing...\")\n\nclient.files.upload(\n local_path=EVAL_DATASET_PATH,\n remote_path=\"\",\n fileset=EVAL_DATASET_NAME,\n workspace=WORKSPACE,\n)\n\nprint(\"\\nEvaluation data files:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=EVAL_DATASET_NAME, workspace=WORKSPACE).data],\n indent=2,\n))", "language": "python", - "source_html": "try:\n client.files.filesets.create(\n name=EVAL_DATASET_NAME,\n description="synthetic tool-calling evaluation data (messages, tools, ground-truth tool_calls)",\n )\n print(f"Created fileset: {EVAL_DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{EVAL_DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=EVAL_DATASET_PATH,\n remote_path="",\n fileset=EVAL_DATASET_NAME,\n)\n\nprint("\\nEvaluation data files:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=EVAL_DATASET_NAME).data],\n indent=2,\n))\n" + "source_html": "try:\n client.files.filesets.create(\n workspace=WORKSPACE,\n name=EVAL_DATASET_NAME,\n description="synthetic tool-calling evaluation data (messages, tools, ground-truth tool_calls)",\n )\n print(f"Created fileset: {EVAL_DATASET_NAME}")\nexcept ConflictError:\n print(f"Fileset '{EVAL_DATASET_NAME}' already exists, continuing...")\n\nclient.files.upload(\n local_path=EVAL_DATASET_PATH,\n remote_path="",\n fileset=EVAL_DATASET_NAME,\n workspace=WORKSPACE,\n)\n\nprint("\\nEvaluation data files:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=EVAL_DATASET_NAME, workspace=WORKSPACE).data],\n indent=2,\n))\n" }, { "type": "markdown", @@ -163,9 +165,9 @@ export default { cells: [ }, { "type": "code", - "source": "HF_TOKEN = os.environ.get(\"HF_TOKEN\")\nif HF_TOKEN is None:\n raise ValueError(\"HF_TOKEN is not set.\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n name=name,\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name)\n\n\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(hf_secret.model_dump_json(indent=2))", + "source": "HF_TOKEN = os.environ.get(\"HF_TOKEN\")\nif HF_TOKEN is None:\n raise ValueError(\"HF_TOKEN is not set.\")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f\"{label} is not set\")\n try:\n secret = client.secrets.create(\n workspace=WORKSPACE,\n name=name,\n value=value,\n )\n print(f\"Created secret: {name}\")\n return secret\n except ConflictError:\n print(f\"Secret '{name}' already exists, continuing...\")\n return client.secrets.retrieve(name=name, workspace=WORKSPACE)\n\n\nhf_secret = create_or_get_secret(\"hf-token\", HF_TOKEN, \"HF_TOKEN\")\nprint(hf_secret.model_dump_json(indent=2))", "language": "python", - "source_html": "HF_TOKEN = os.environ.get("HF_TOKEN")\nif HF_TOKEN is None:\n raise ValueError("HF_TOKEN is not set.")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f"{label} is not set")\n try:\n secret = client.secrets.create(\n name=name,\n value=value,\n )\n print(f"Created secret: {name}")\n return secret\n except ConflictError:\n print(f"Secret '{name}' already exists, continuing...")\n return client.secrets.retrieve(name=name)\n\n\nhf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN")\nprint(hf_secret.model_dump_json(indent=2))\n" + "source_html": "HF_TOKEN = os.environ.get("HF_TOKEN")\nif HF_TOKEN is None:\n raise ValueError("HF_TOKEN is not set.")\n\n\ndef create_or_get_secret(name: str, value: str | None, label: str):\n if not value:\n raise ValueError(f"{label} is not set")\n try:\n secret = client.secrets.create(\n workspace=WORKSPACE,\n name=name,\n value=value,\n )\n print(f"Created secret: {name}")\n return secret\n except ConflictError:\n print(f"Secret '{name}' already exists, continuing...")\n return client.secrets.retrieve(name=name, workspace=WORKSPACE)\n\n\nhf_secret = create_or_get_secret("hf-token", HF_TOKEN, "HF_TOKEN")\nprint(hf_secret.model_dump_json(indent=2))\n" }, { "type": "markdown", @@ -174,20 +176,20 @@ export default { cells: [ }, { "type": "code", - "source": "import time\n\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = \"meta-llama/Llama-3.2-1B-Instruct\"\nMODEL_NAME = \"llama-3-2-1b-base\"\n\ntry:\n base_model_fs = client.files.filesets.create(\n name=MODEL_NAME,\n description=\"Llama 3.2 1B Instruct base model from HuggingFace\",\n storage=HuggingfaceStorageConfigParam(\n type=\"huggingface\",\n repo_id=HF_REPO_ID,\n repo_type=\"model\",\n token_secret=hf_secret.name,\n ),\n metadata={\n \"model\": {\n \"tool_calling\": {\n \"tool_call_parser\": \"llama3_json\",\n \"auto_tool_choice\": True,\n }\n }\n },\n )\n print(f\"Created base model fileset: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model fileset already exists. Skipping creation.\")\n base_model_fs = client.files.filesets.retrieve(\n name=MODEL_NAME,\n )\n\ntry:\n base_model = client.models.create(\n name=MODEL_NAME,\n fileset=f\"{WORKSPACE}/{MODEL_NAME}\",\n )\n print(f\"Created Model Entity: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model already exists. Updating fileset if different.\")\n base_model = client.models.update(\n name=MODEL_NAME,\n fileset=f\"{WORKSPACE}/{MODEL_NAME}\",\n )\n\nprint(f\"\\nBase model fileset: fileset://{WORKSPACE}/{base_model.name}\")\nprint(\"Base model fileset files list:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=MODEL_NAME).data],\n indent=2,\n))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint(\"\\nWaiting for ModelSpec to be populated...\")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n time.sleep(2)\n base_model = client.models.retrieve(\n name=MODEL_NAME,\n )\n\nprint(f\"ModelSpec populated: {base_model.spec.model_dump()}\")", + "source": "import time\n\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = \"meta-llama/Llama-3.2-1B-Instruct\"\nMODEL_NAME = \"llama-3-2-1b-base\"\n\ntry:\n base_model_fs = client.files.filesets.create(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n description=\"Llama 3.2 1B Instruct base model from HuggingFace\",\n storage=HuggingfaceStorageConfigParam(\n type=\"huggingface\",\n repo_id=HF_REPO_ID,\n repo_type=\"model\",\n token_secret=hf_secret.name,\n ),\n metadata={\n \"model\": {\n \"tool_calling\": {\n \"tool_call_parser\": \"llama3_json\",\n \"auto_tool_choice\": True,\n }\n }\n },\n )\n print(f\"Created base model fileset: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model fileset already exists. Skipping creation.\")\n base_model_fs = client.files.filesets.retrieve(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n )\n\ntry:\n base_model = client.models.create(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n fileset=f\"{WORKSPACE}/{MODEL_NAME}\",\n )\n print(f\"Created Model Entity: {MODEL_NAME}\")\nexcept ConflictError:\n print(f\"Base model already exists. Updating fileset if different.\")\n base_model = client.models.update(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n fileset=f\"{WORKSPACE}/{MODEL_NAME}\",\n )\n\nprint(f\"\\nBase model fileset: fileset://{WORKSPACE}/{base_model.name}\")\nprint(\"Base model fileset files list:\")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=WORKSPACE).data],\n indent=2,\n))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint(\"\\nWaiting for ModelSpec to be populated...\")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f\"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds\")\n time.sleep(2)\n base_model = client.models.retrieve(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n )\n\nprint(f\"ModelSpec populated: {base_model.spec.model_dump()}\")", "language": "python", - "source_html": "import time\n\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = "meta-llama/Llama-3.2-1B-Instruct"\nMODEL_NAME = "llama-3-2-1b-base"\n\ntry:\n base_model_fs = client.files.filesets.create(\n name=MODEL_NAME,\n description="Llama 3.2 1B Instruct base model from HuggingFace",\n storage=HuggingfaceStorageConfigParam(\n type="huggingface",\n repo_id=HF_REPO_ID,\n repo_type="model",\n token_secret=hf_secret.name,\n ),\n metadata={\n "model": {\n "tool_calling": {\n "tool_call_parser": "llama3_json",\n "auto_tool_choice": True,\n }\n }\n },\n )\n print(f"Created base model fileset: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model fileset already exists. Skipping creation.")\n base_model_fs = client.files.filesets.retrieve(\n name=MODEL_NAME,\n )\n\ntry:\n base_model = client.models.create(\n name=MODEL_NAME,\n fileset=f"{WORKSPACE}/{MODEL_NAME}",\n )\n print(f"Created Model Entity: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model already exists. Updating fileset if different.")\n base_model = client.models.update(\n name=MODEL_NAME,\n fileset=f"{WORKSPACE}/{MODEL_NAME}",\n )\n\nprint(f"\\nBase model fileset: fileset://{WORKSPACE}/{base_model.name}")\nprint("Base model fileset files list:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=MODEL_NAME).data],\n indent=2,\n))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint("\\nWaiting for ModelSpec to be populated...")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds")\n time.sleep(2)\n base_model = client.models.retrieve(\n name=MODEL_NAME,\n )\n\nprint(f"ModelSpec populated: {base_model.spec.model_dump()}")\n" + "source_html": "import time\n\nfrom nemo_platform.types.files import HuggingfaceStorageConfigParam\n\nHF_REPO_ID = "meta-llama/Llama-3.2-1B-Instruct"\nMODEL_NAME = "llama-3-2-1b-base"\n\ntry:\n base_model_fs = client.files.filesets.create(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n description="Llama 3.2 1B Instruct base model from HuggingFace",\n storage=HuggingfaceStorageConfigParam(\n type="huggingface",\n repo_id=HF_REPO_ID,\n repo_type="model",\n token_secret=hf_secret.name,\n ),\n metadata={\n "model": {\n "tool_calling": {\n "tool_call_parser": "llama3_json",\n "auto_tool_choice": True,\n }\n }\n },\n )\n print(f"Created base model fileset: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model fileset already exists. Skipping creation.")\n base_model_fs = client.files.filesets.retrieve(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n )\n\ntry:\n base_model = client.models.create(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n fileset=f"{WORKSPACE}/{MODEL_NAME}",\n )\n print(f"Created Model Entity: {MODEL_NAME}")\nexcept ConflictError:\n print(f"Base model already exists. Updating fileset if different.")\n base_model = client.models.update(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n fileset=f"{WORKSPACE}/{MODEL_NAME}",\n )\n\nprint(f"\\nBase model fileset: fileset://{WORKSPACE}/{base_model.name}")\nprint("Base model fileset files list:")\nprint(json.dumps(\n [f.model_dump() for f in client.files.list(fileset=MODEL_NAME, workspace=WORKSPACE).data],\n indent=2,\n))\n\n# Wait for ModelSpec to be populated from the checkpoint\nprint("\\nWaiting for ModelSpec to be populated...")\nSPEC_TIMEOUT_SECONDS = 120\nspec_start = time.time()\nwhile not base_model.spec:\n if time.time() - spec_start > SPEC_TIMEOUT_SECONDS:\n raise TimeoutError(f"ModelSpec not populated within {SPEC_TIMEOUT_SECONDS} seconds")\n time.sleep(2)\n base_model = client.models.retrieve(\n workspace=WORKSPACE,\n name=MODEL_NAME,\n )\n\nprint(f"ModelSpec populated: {base_model.spec.model_dump()}")\n" }, { "type": "markdown", - "source": "## 4. Create LoRA Fine-Tuning Job\n\nCreate a customization job using SFT training with LoRA PEFT. The `peft=LoRaParamsParam()` parameter enables LoRA instead of full-weight fine-tuning.\n\n**LoRA defaults** (can be overridden in `LoRaParamsParam()`):\n- `rank`: LoRA rank (dimensionality of the low-rank matrices)\n- `alpha`: Scaling factor for LoRA updates\n- `target_modules`: Which model layers to apply LoRA to", - "source_html": "

4. Create LoRA Fine-Tuning Job

\n

Create a customization job using SFT training with LoRA PEFT. The peft=LoRaParamsParam() parameter enables LoRA instead of full-weight fine-tuning.

\n

LoRA defaults (can be overridden in LoRaParamsParam()):

\n
    \n
  • rank: LoRA rank (dimensionality of the low-rank matrices)
  • \n
  • alpha: Scaling factor for LoRA updates
  • \n
  • target_modules: Which model layers to apply LoRA to
  • \n
\n" + "source": "## 4. Create LoRA Fine-Tuning Job\n\nCreate an **Automodel** customization job using `AutomodelJobInput` with `finetuning_type: lora`. After training completes, deploy the base model with LoRA support in the next section.", + "source_html": "

4. Create LoRA Fine-Tuning Job

\n

Create an Automodel customization job using AutomodelJobInput with finetuning_type: lora. After training completes, deploy the base model with LoRA support in the next section.

\n" }, { "type": "code", - "source": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n LoRaParamsParam,\n DeploymentParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"tool-calling-lora-{job_suffix}\"\n\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n spec=CustomizationJobInputParam(\n model=f\"{WORKSPACE}/{base_model.name}\",\n dataset=f\"fileset://{WORKSPACE}/{DATASET_NAME}\",\n training=SftTrainingParam(\n type=\"sft\",\n epochs=4,\n batch_size=4,\n learning_rate=0.0001,\n max_seq_length=2048,\n micro_batch_size=1,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n deployment_config=DeploymentParamsParam(\n lora_enabled=True,\n ),\n ),\n)\n\nprint(job.model_dump_json(indent=2))", + "source": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f\"tool-calling-lora-{job_suffix}\"\nOUTPUT_NAME = f\"tool-calling-adapter-{job_suffix}\"\n\nspec = AutomodelJobInput(\n model=f\"{WORKSPACE}/{base_model.name}\",\n dataset={\"training\": f\"{WORKSPACE}/{DATASET_NAME}\"},\n training={\n \"training_type\": \"sft\",\n \"finetuning_type\": \"lora\",\n \"max_seq_length\": 2048,\n },\n schedule={\"epochs\": 4},\n batch={\"global_batch_size\": 4, \"micro_batch_size\": 1},\n optimizer={\"learning_rate\": 1e-4},\n parallelism={\"num_gpus_per_node\": 1},\n output={\"name\": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace=WORKSPACE, name=JOB_NAME\n)\n\nprint(f\"Submitted job: {job.job.name}\")\nprint(f\"Output adapter: {OUTPUT_NAME}\")", "language": "python", - "source_html": "import uuid\nfrom nemo_platform.types.customization import (\n CustomizationJobInputParam,\n SftTrainingParam,\n ParallelismParamsParam,\n LoRaParamsParam,\n DeploymentParamsParam,\n)\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"tool-calling-lora-{job_suffix}"\n\n\njob = client.customization.jobs.create(\n name=JOB_NAME,\n spec=CustomizationJobInputParam(\n model=f"{WORKSPACE}/{base_model.name}",\n dataset=f"fileset://{WORKSPACE}/{DATASET_NAME}",\n training=SftTrainingParam(\n type="sft",\n epochs=4,\n batch_size=4,\n learning_rate=0.0001,\n max_seq_length=2048,\n micro_batch_size=1,\n peft=LoRaParamsParam(),\n parallelism=ParallelismParamsParam(\n num_gpus_per_node=1,\n num_nodes=1,\n tensor_parallel_size=1,\n pipeline_parallel_size=1,\n ),\n ),\n deployment_config=DeploymentParamsParam(\n lora_enabled=True,\n ),\n ),\n)\n\nprint(job.model_dump_json(indent=2))\n" + "source_html": "import uuid\nfrom nemo_automodel_plugin.schema import AutomodelJobInput\n\njob_suffix = uuid.uuid4().hex[:4]\nJOB_NAME = f"tool-calling-lora-{job_suffix}"\nOUTPUT_NAME = f"tool-calling-adapter-{job_suffix}"\n\nspec = AutomodelJobInput(\n model=f"{WORKSPACE}/{base_model.name}",\n dataset={"training": f"{WORKSPACE}/{DATASET_NAME}"},\n training={\n "training_type": "sft",\n "finetuning_type": "lora",\n "max_seq_length": 2048,\n },\n schedule={"epochs": 4},\n batch={"global_batch_size": 4, "micro_batch_size": 1},\n optimizer={"learning_rate": 1e-4},\n parallelism={"num_gpus_per_node": 1},\n output={"name": OUTPUT_NAME},\n)\n\njob = client.customization.automodel.jobs.create(\n spec=spec, workspace=WORKSPACE, name=JOB_NAME\n)\n\nprint(f"Submitted job: {job.job.name}")\nprint(f"Output adapter: {OUTPUT_NAME}")\n" }, { "type": "markdown", @@ -196,26 +198,26 @@ export default { cells: [ }, { "type": "code", - "source": "import time\nfrom IPython.display import clear_output\n\nTERMINAL_JOB_STATUSES = {\"completed\", \"cancelled\", \"error\"}\n\n\ndef wait_for_job(poll_fn, label, timeout_minutes=60, poll_interval=10, display_fn=None):\n \"\"\"Poll a platform job until it reaches a terminal state.\n\n Both Customizer and Evaluator use the Core Jobs service, so the same\n PlatformJobStatus values apply: created, pending, active, cancelled,\n cancelling, error, completed, paused, pausing, resuming.\n\n Args:\n poll_fn: Callable returning a status object with .name and .status attributes.\n label: Display label for progress output.\n timeout_minutes: Maximum time to wait before returning.\n poll_interval: Seconds between polls.\n display_fn: Optional callable(status) to print extra details after the header.\n\n Returns:\n The final status object.\n \"\"\"\n start = time.time()\n timeout = timeout_minutes * 60\n\n while True:\n status = poll_fn()\n elapsed = time.time() - start\n elapsed_min, elapsed_sec = divmod(int(elapsed), 60)\n\n clear_output(wait=True)\n print(f\"[{label}] Job: {status.name}\")\n print(f\"[{label}] Status: {status.status}\")\n print(f\"[{label}] Elapsed: {elapsed_min}m {elapsed_sec}s\")\n\n if display_fn:\n display_fn(status)\n\n if status.status in TERMINAL_JOB_STATUSES:\n print(f\"\\n[{label}] Job finished: {status.status}\")\n return status\n\n if elapsed > timeout:\n print(f\"\\n[{label}] Timeout after {timeout_minutes} minutes\")\n return status\n\n time.sleep(poll_interval)\n\n\ndef training_progress(status):\n \"\"\"Extract and display training step progress.\"\"\"\n for step in status.steps or []:\n if step.name == \"customization-training-job\":\n for task in step.tasks or []:\n details = task.status_details or {}\n s, mx = details.get(\"step\"), details.get(\"max_steps\")\n if s is not None and mx is not None:\n print(f\"Training: Step {s}/{mx} ({(s / mx) * 100:.1f}%)\")\n if phase := details.get(\"phase\"):\n print(f\"Phase: {phase}\")\n return\n print(\"Training step not started yet\")\n\nprint(\"Defined wait_for_job helper function\")", + "source": "import time\nfrom IPython.display import clear_output\n\nTERMINAL_JOB_STATUSES = {\"completed\", \"failed\", \"cancelled\", \"error\"}\n\n\ndef wait_for_job(poll_fn, label, timeout_minutes=60, poll_interval=10, display_fn=None):\n \"\"\"Poll a platform job until it reaches a terminal state.\n\n Both Customizer and Evaluator use the Core Jobs service, so the same\n PlatformJobStatus values apply: created, pending, active, cancelled,\n cancelling, error, completed, paused, pausing, resuming.\n\n Args:\n poll_fn: Callable returning a status object with .name and .status attributes.\n label: Display label for progress output.\n timeout_minutes: Maximum time to wait before returning.\n poll_interval: Seconds between polls.\n display_fn: Optional callable(status) to print extra details after the header.\n\n Returns:\n The final status object.\n \"\"\"\n start = time.time()\n timeout = timeout_minutes * 60\n\n while True:\n status = poll_fn()\n elapsed = time.time() - start\n elapsed_min, elapsed_sec = divmod(int(elapsed), 60)\n\n clear_output(wait=True)\n print(f\"[{label}] Job: {status.name}\")\n print(f\"[{label}] Status: {status.status}\")\n print(f\"[{label}] Elapsed: {elapsed_min}m {elapsed_sec}s\")\n\n if display_fn:\n display_fn(status)\n\n if status.status in TERMINAL_JOB_STATUSES:\n print(f\"\\n[{label}] Job finished: {status.status}\")\n return status\n\n if elapsed > timeout:\n print(f\"\\n[{label}] Timeout after {timeout_minutes} minutes\")\n return status\n\n time.sleep(poll_interval)\n\n\ndef training_progress(status):\n \"\"\"Extract and display training step progress.\"\"\"\n for step in status.steps or []:\n if step.name == \"training\":\n for task in step.tasks or []:\n details = task.status_details or {}\n s, mx = details.get(\"step\"), details.get(\"max_steps\")\n if s is not None and mx is not None:\n print(f\"Training: Step {s}/{mx} ({(s / mx) * 100:.1f}%)\")\n if phase := details.get(\"phase\"):\n print(f\"Phase: {phase}\")\n return\n print(\"Training step not started yet\")\n\nprint(\"Defined wait_for_job helper function\")", "language": "python", - "source_html": "import time\nfrom IPython.display import clear_output\n\nTERMINAL_JOB_STATUSES = {"completed", "cancelled", "error"}\n\n\ndef wait_for_job(poll_fn, label, timeout_minutes=60, poll_interval=10, display_fn=None):\n """Poll a platform job until it reaches a terminal state.\n\n Both Customizer and Evaluator use the Core Jobs service, so the same\n PlatformJobStatus values apply: created, pending, active, cancelled,\n cancelling, error, completed, paused, pausing, resuming.\n\n Args:\n poll_fn: Callable returning a status object with .name and .status attributes.\n label: Display label for progress output.\n timeout_minutes: Maximum time to wait before returning.\n poll_interval: Seconds between polls.\n display_fn: Optional callable(status) to print extra details after the header.\n\n Returns:\n The final status object.\n """\n start = time.time()\n timeout = timeout_minutes * 60\n\n while True:\n status = poll_fn()\n elapsed = time.time() - start\n elapsed_min, elapsed_sec = divmod(int(elapsed), 60)\n\n clear_output(wait=True)\n print(f"[{label}] Job: {status.name}")\n print(f"[{label}] Status: {status.status}")\n print(f"[{label}] Elapsed: {elapsed_min}m {elapsed_sec}s")\n\n if display_fn:\n display_fn(status)\n\n if status.status in TERMINAL_JOB_STATUSES:\n print(f"\\n[{label}] Job finished: {status.status}")\n return status\n\n if elapsed > timeout:\n print(f"\\n[{label}] Timeout after {timeout_minutes} minutes")\n return status\n\n time.sleep(poll_interval)\n\n\ndef training_progress(status):\n """Extract and display training step progress."""\n for step in status.steps or []:\n if step.name == "customization-training-job":\n for task in step.tasks or []:\n details = task.status_details or {}\n s, mx = details.get("step"), details.get("max_steps")\n if s is not None and mx is not None:\n print(f"Training: Step {s}/{mx} ({(s / mx) * 100:.1f}%)")\n if phase := details.get("phase"):\n print(f"Phase: {phase}")\n return\n print("Training step not started yet")\n\nprint("Defined wait_for_job helper function")\n" + "source_html": "import time\nfrom IPython.display import clear_output\n\nTERMINAL_JOB_STATUSES = {"completed", "failed", "cancelled", "error"}\n\n\ndef wait_for_job(poll_fn, label, timeout_minutes=60, poll_interval=10, display_fn=None):\n """Poll a platform job until it reaches a terminal state.\n\n Both Customizer and Evaluator use the Core Jobs service, so the same\n PlatformJobStatus values apply: created, pending, active, cancelled,\n cancelling, error, completed, paused, pausing, resuming.\n\n Args:\n poll_fn: Callable returning a status object with .name and .status attributes.\n label: Display label for progress output.\n timeout_minutes: Maximum time to wait before returning.\n poll_interval: Seconds between polls.\n display_fn: Optional callable(status) to print extra details after the header.\n\n Returns:\n The final status object.\n """\n start = time.time()\n timeout = timeout_minutes * 60\n\n while True:\n status = poll_fn()\n elapsed = time.time() - start\n elapsed_min, elapsed_sec = divmod(int(elapsed), 60)\n\n clear_output(wait=True)\n print(f"[{label}] Job: {status.name}")\n print(f"[{label}] Status: {status.status}")\n print(f"[{label}] Elapsed: {elapsed_min}m {elapsed_sec}s")\n\n if display_fn:\n display_fn(status)\n\n if status.status in TERMINAL_JOB_STATUSES:\n print(f"\\n[{label}] Job finished: {status.status}")\n return status\n\n if elapsed > timeout:\n print(f"\\n[{label}] Timeout after {timeout_minutes} minutes")\n return status\n\n time.sleep(poll_interval)\n\n\ndef training_progress(status):\n """Extract and display training step progress."""\n for step in status.steps or []:\n if step.name == "training":\n for task in step.tasks or []:\n details = task.status_details or {}\n s, mx = details.get("step"), details.get("max_steps")\n if s is not None and mx is not None:\n print(f"Training: Step {s}/{mx} ({(s / mx) * 100:.1f}%)")\n if phase := details.get("phase"):\n print(f"Phase: {phase}")\n return\n print("Training step not started yet")\n\nprint("Defined wait_for_job helper function")\n" }, { "type": "code", - "source": "job_status = wait_for_job(\n poll_fn=lambda: client.customization.jobs.get_status(name=job.name),\n label=\"Training\",\n timeout_minutes=120,\n display_fn=training_progress,\n)", + "source": "job_status = wait_for_job(\n poll_fn=lambda: client.jobs.get_status(name=job.job.name, workspace=WORKSPACE),\n label=\"Training\",\n timeout_minutes=120,\n display_fn=training_progress,\n)\n\nif job_status.status != \"completed\":\n raise RuntimeError(f\"Training job finished with status: {job_status.status}\")", "language": "python", - "source_html": "job_status = wait_for_job(\n poll_fn=lambda: client.customization.jobs.get_status(name=job.name),\n label="Training",\n timeout_minutes=120,\n display_fn=training_progress,\n)\n" + "source_html": "job_status = wait_for_job(\n poll_fn=lambda: client.jobs.get_status(name=job.job.name, workspace=WORKSPACE),\n label="Training",\n timeout_minutes=120,\n display_fn=training_progress,\n)\n\nif job_status.status != "completed":\n raise RuntimeError(f"Training job finished with status: {job_status.status}")\n" }, { "type": "markdown", - "source": "## 5. Verify Auto-Deployed Model\n\nSince we set `lora_enabled=True` in the customization job's `deployment_config`, the platform automatically creates a NIM deployment for the base model after training completes. The LoRA adapter is attached to the base model entity (enabled by default) and the deployment serves both the base weights and the adapter through a single NIM instance.", - "source_html": "

5. Verify Auto-Deployed Model

\n

Since we set lora_enabled=True in the customization job's deployment_config, the platform automatically creates a NIM deployment for the base model after training completes. The LoRA adapter is attached to the base model entity (enabled by default) and the deployment serves both the base weights and the adapter through a single NIM instance.

\n" + "source": "## 5. Deploy Fine-Tuned Model\n\nAfter training completes, verify the LoRA adapter is attached to the base model entity, then create a NIM deployment with `lora_enabled=True` so both the base weights and adapter are served through a single deployment.", + "source_html": "

5. Deploy Fine-Tuned Model

\n

After training completes, verify the LoRA adapter is attached to the base model entity, then create a NIM deployment with lora_enabled=True so both the base weights and adapter are served through a single deployment.

\n" }, { "type": "code", - "source": "ADAPTER_NAME = job.spec.output.name\nprint(f\"Looking for adapter: {ADAPTER_NAME}\")\n\n# The adapter may not be attached to the model entity immediately after\n# training completes — poll until it appears.\nADAPTER_TIMEOUT = 120\nadapter_start = time.time()\nadapter = None\nwhile time.time() - adapter_start < ADAPTER_TIMEOUT:\n base_model = client.models.retrieve(name=MODEL_NAME)\n matches = [a for a in (base_model.adapters or []) if a.name == ADAPTER_NAME]\n if matches:\n adapter = matches[0]\n break\n print(f\"Adapter not yet attached, retrying... ({int(time.time() - adapter_start)}s)\")\n time.sleep(5)\n\nif adapter is None:\n raise TimeoutError(\n f\"Adapter '{ADAPTER_NAME}' not found on model '{MODEL_NAME}' within {ADAPTER_TIMEOUT}s\"\n )\n\nprint(f\"Base model: {base_model.name}\")\nprint(f\"Adapter:\\n{adapter.model_dump_json(indent=2)}\")", + "source": "ADAPTER_NAME = OUTPUT_NAME\nprint(f\"Looking for adapter: {ADAPTER_NAME}\")\n\n# The adapter may not be attached to the model entity immediately after\n# training completes — poll until it appears.\nADAPTER_TIMEOUT = 120\nadapter_start = time.time()\nadapter = None\nwhile time.time() - adapter_start < ADAPTER_TIMEOUT:\n base_model = client.models.retrieve(name=MODEL_NAME, workspace=WORKSPACE)\n matches = [a for a in (base_model.adapters or []) if a.name == ADAPTER_NAME]\n if matches:\n adapter = matches[0]\n break\n print(f\"Adapter not yet attached, retrying... ({int(time.time() - adapter_start)}s)\")\n time.sleep(5)\n\nif adapter is None:\n raise TimeoutError(\n f\"Adapter '{ADAPTER_NAME}' not found on model '{MODEL_NAME}' within {ADAPTER_TIMEOUT}s\"\n )\n\nprint(f\"Base model: {base_model.name}\")\nprint(f\"Adapter:\\n{adapter.model_dump_json(indent=2)}\")\n\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f\"tool-calling-deploy-cfg-{deploy_suffix}\"\ndeployment_name = f\"tool-calling-deploy-{deploy_suffix}\"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=WORKSPACE,\n name=DEPLOYMENT_CONFIG_NAME,\n engine=\"vllm\",\n model_spec={\n \"model_namespace\": WORKSPACE,\n \"model_name\": MODEL_NAME,\n \"lora_enabled\": True,\n },\n executor_config={\n \"gpu\": 1,\n \"image_name\": \"vllm/vllm-openai\",\n \"image_tag\": \"v0.22.1\",\n \"additional_args\": [\"--max-lora-rank\", \"32\"],\n },\n)\n\ndeployment = client.inference.deployments.create(\n workspace=WORKSPACE,\n name=deployment_name,\n config=deployment_config.name,\n)\n\nprint(f\"Deployment status: {deployment.status}\")", "language": "python", - "source_html": "ADAPTER_NAME = job.spec.output.name\nprint(f"Looking for adapter: {ADAPTER_NAME}")\n\n# The adapter may not be attached to the model entity immediately after\n# training completes — poll until it appears.\nADAPTER_TIMEOUT = 120\nadapter_start = time.time()\nadapter = None\nwhile time.time() - adapter_start < ADAPTER_TIMEOUT:\n base_model = client.models.retrieve(name=MODEL_NAME)\n matches = [a for a in (base_model.adapters or []) if a.name == ADAPTER_NAME]\n if matches:\n adapter = matches[0]\n break\n print(f"Adapter not yet attached, retrying... ({int(time.time() - adapter_start)}s)")\n time.sleep(5)\n\nif adapter is None:\n raise TimeoutError(\n f"Adapter '{ADAPTER_NAME}' not found on model '{MODEL_NAME}' within {ADAPTER_TIMEOUT}s"\n )\n\nprint(f"Base model: {base_model.name}")\nprint(f"Adapter:\\n{adapter.model_dump_json(indent=2)}")\n" + "source_html": "ADAPTER_NAME = OUTPUT_NAME\nprint(f"Looking for adapter: {ADAPTER_NAME}")\n\n# The adapter may not be attached to the model entity immediately after\n# training completes — poll until it appears.\nADAPTER_TIMEOUT = 120\nadapter_start = time.time()\nadapter = None\nwhile time.time() - adapter_start < ADAPTER_TIMEOUT:\n base_model = client.models.retrieve(name=MODEL_NAME, workspace=WORKSPACE)\n matches = [a for a in (base_model.adapters or []) if a.name == ADAPTER_NAME]\n if matches:\n adapter = matches[0]\n break\n print(f"Adapter not yet attached, retrying... ({int(time.time() - adapter_start)}s)")\n time.sleep(5)\n\nif adapter is None:\n raise TimeoutError(\n f"Adapter '{ADAPTER_NAME}' not found on model '{MODEL_NAME}' within {ADAPTER_TIMEOUT}s"\n )\n\nprint(f"Base model: {base_model.name}")\nprint(f"Adapter:\\n{adapter.model_dump_json(indent=2)}")\n\ndeploy_suffix = uuid.uuid4().hex[:4]\nDEPLOYMENT_CONFIG_NAME = f"tool-calling-deploy-cfg-{deploy_suffix}"\ndeployment_name = f"tool-calling-deploy-{deploy_suffix}"\n\ndeployment_config = client.inference.deployment_configs.create(\n workspace=WORKSPACE,\n name=DEPLOYMENT_CONFIG_NAME,\n engine="vllm",\n model_spec={\n "model_namespace": WORKSPACE,\n "model_name": MODEL_NAME,\n "lora_enabled": True,\n },\n executor_config={\n "gpu": 1,\n "image_name": "vllm/vllm-openai",\n "image_tag": "v0.22.1",\n "additional_args": ["--max-lora-rank", "32"],\n },\n)\n\ndeployment = client.inference.deployments.create(\n workspace=WORKSPACE,\n name=deployment_name,\n config=deployment_config.name,\n)\n\nprint(f"Deployment status: {deployment.status}")\n" }, { "type": "markdown", @@ -224,9 +226,9 @@ export default { cells: [ }, { "type": "code", - "source": "DEPLOYMENT_NAME = f\"sft-deploy-{MODEL_NAME}\"\n\nTIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f\"Monitoring deployment '{DEPLOYMENT_NAME}'...\")\nprint(f\"Timeout: {TIMEOUT_MINUTES} minutes\\n\")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=DEPLOYMENT_NAME,\n )\n\n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n\n clear_output(wait=True)\n print(f\"Deployment: {DEPLOYMENT_NAME}\")\n print(f\"Status: {deployment_status.status}\")\n print(f\"Elapsed time: {elapsed_min}m {elapsed_sec}s\")\n\n if deployment_status.status == \"READY\":\n print(\"\\nDeployment is ready!\")\n if not client.models.wait_for_gateway(DEPLOYMENT_NAME, workspace=WORKSPACE, timeout=60):\n raise RuntimeError(\"Inference gateway did not become ready\")\n break\n\n if deployment_status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\", \"DELETED\"):\n raise RuntimeError(f\"Deployment failed with status: {deployment_status.status}\")\n\n if elapsed > timeout_seconds:\n raise TimeoutError(f\"Deployment timeout after {TIMEOUT_MINUTES} minutes\")\n\n time.sleep(15)", + "source": "TIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f\"Monitoring deployment '{deployment_name}'...\")\nprint(f\"Timeout: {TIMEOUT_MINUTES} minutes\\n\")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=deployment_name,\n workspace=WORKSPACE,\n )\n\n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n\n clear_output(wait=True)\n print(f\"Deployment: {deployment_name}\")\n print(f\"Status: {deployment_status.status}\")\n print(f\"Elapsed time: {elapsed_min}m {elapsed_sec}s\")\n\n if deployment_status.status in (\"RUNNING\", \"READY\"):\n print(\"\\nDeployment is ready!\")\n if not client.models.wait_for_gateway(deployment_name, workspace=WORKSPACE, timeout=60):\n raise RuntimeError(\"Inference gateway did not become ready\")\n break\n\n if deployment_status.status in (\"FAILED\", \"ERROR\", \"TERMINATED\", \"LOST\", \"DELETED\"):\n raise RuntimeError(f\"Deployment failed with status: {deployment_status.status}\")\n\n if elapsed > timeout_seconds:\n raise TimeoutError(f\"Deployment timeout after {TIMEOUT_MINUTES} minutes\")\n\n time.sleep(15)", "language": "python", - "source_html": "DEPLOYMENT_NAME = f"sft-deploy-{MODEL_NAME}"\n\nTIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f"Monitoring deployment '{DEPLOYMENT_NAME}'...")\nprint(f"Timeout: {TIMEOUT_MINUTES} minutes\\n")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=DEPLOYMENT_NAME,\n )\n\n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n\n clear_output(wait=True)\n print(f"Deployment: {DEPLOYMENT_NAME}")\n print(f"Status: {deployment_status.status}")\n print(f"Elapsed time: {elapsed_min}m {elapsed_sec}s")\n\n if deployment_status.status == "READY":\n print("\\nDeployment is ready!")\n if not client.models.wait_for_gateway(DEPLOYMENT_NAME, workspace=WORKSPACE, timeout=60):\n raise RuntimeError("Inference gateway did not become ready")\n break\n\n if deployment_status.status in ("FAILED", "ERROR", "TERMINATED", "LOST", "DELETED"):\n raise RuntimeError(f"Deployment failed with status: {deployment_status.status}")\n\n if elapsed > timeout_seconds:\n raise TimeoutError(f"Deployment timeout after {TIMEOUT_MINUTES} minutes")\n\n time.sleep(15)\n" + "source_html": "TIMEOUT_MINUTES = 30\nstart_time = time.time()\ntimeout_seconds = TIMEOUT_MINUTES * 60\n\nprint(f"Monitoring deployment '{deployment_name}'...")\nprint(f"Timeout: {TIMEOUT_MINUTES} minutes\\n")\n\nwhile True:\n deployment_status = client.inference.deployments.retrieve(\n name=deployment_name,\n workspace=WORKSPACE,\n )\n\n elapsed = time.time() - start_time\n elapsed_min = int(elapsed // 60)\n elapsed_sec = int(elapsed % 60)\n\n clear_output(wait=True)\n print(f"Deployment: {deployment_name}")\n print(f"Status: {deployment_status.status}")\n print(f"Elapsed time: {elapsed_min}m {elapsed_sec}s")\n\n if deployment_status.status in ("RUNNING", "READY"):\n print("\\nDeployment is ready!")\n if not client.models.wait_for_gateway(deployment_name, workspace=WORKSPACE, timeout=60):\n raise RuntimeError("Inference gateway did not become ready")\n break\n\n if deployment_status.status in ("FAILED", "ERROR", "TERMINATED", "LOST", "DELETED"):\n raise RuntimeError(f"Deployment failed with status: {deployment_status.status}")\n\n if elapsed > timeout_seconds:\n raise TimeoutError(f"Deployment timeout after {TIMEOUT_MINUTES} minutes")\n\n time.sleep(15)\n" }, { "type": "markdown", @@ -235,9 +237,9 @@ export default { cells: [ }, { "type": "code", - "source": "test_messages = [\n {\"role\": \"user\", \"content\": \"Calculate the factorial of 12 using math functions.\"},\n]\n\ntest_tools = [\n {\n \"type\": \"function\",\n \"function\": {\n \"name\": \"math_factorial\",\n \"description\": \"Calculate the factorial of a given number.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"number\": {\n \"type\": \"integer\",\n \"description\": \"The number for which factorial needs to be calculated.\",\n }\n },\n \"required\": [\"number\"],\n },\n },\n }\n]\n\n\ndef test_tool_calling(model_name: str, label: str):\n \"\"\"Send a tool calling request and display the response.\"\"\"\n response = client.inference.gateway.model.post(\n \"v1/chat/completions\",\n name=model_name,\n body={\n \"messages\": test_messages,\n \"tools\": test_tools,\n \"tool_choice\": \"auto\",\n \"temperature\": 0,\n \"max_tokens\": 256,\n },\n )\n\n print(f\"{'=' * 60}\")\n print(f\" {label}\")\n print(f\"{'=' * 60}\")\n print(json.dumps(response, indent=2))\n\n\ntest_tool_calling(MODEL_NAME, \"BASE MODEL (before fine-tuning)\")\ntest_tool_calling(ADAPTER_NAME, \"FINE-TUNED MODEL (after LoRA)\")", + "source": "test_messages = [\n {\"role\": \"user\", \"content\": \"Calculate the factorial of 12 using math functions.\"},\n]\n\ntest_tools = [\n {\n \"type\": \"function\",\n \"function\": {\n \"name\": \"math_factorial\",\n \"description\": \"Calculate the factorial of a given number.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"number\": {\n \"type\": \"integer\",\n \"description\": \"The number for which factorial needs to be calculated.\",\n }\n },\n \"required\": [\"number\"],\n },\n },\n }\n]\n\n\ndef test_tool_calling(model_name: str, label: str):\n \"\"\"Send a tool calling request and display the response.\"\"\"\n response = client.inference.gateway.provider.post(\n \"v1/chat/completions\",\n name=deployment_name,\n workspace=WORKSPACE,\n body={\n \"model\": model_name,\n \"messages\": test_messages,\n \"tools\": test_tools,\n \"tool_choice\": \"auto\",\n \"temperature\": 0,\n \"max_tokens\": 256,\n },\n )\n\n print(f\"{'=' * 60}\")\n print(f\" {label}\")\n print(f\"{'=' * 60}\")\n print(json.dumps(response, indent=2))\n\n\ntest_tool_calling(MODEL_NAME, \"BASE MODEL (before fine-tuning)\")\ntest_tool_calling(ADAPTER_NAME, \"FINE-TUNED MODEL (after LoRA)\")", "language": "python", - "source_html": "test_messages = [\n {"role": "user", "content": "Calculate the factorial of 12 using math functions."},\n]\n\ntest_tools = [\n {\n "type": "function",\n "function": {\n "name": "math_factorial",\n "description": "Calculate the factorial of a given number.",\n "parameters": {\n "type": "object",\n "properties": {\n "number": {\n "type": "integer",\n "description": "The number for which factorial needs to be calculated.",\n }\n },\n "required": ["number"],\n },\n },\n }\n]\n\n\ndef test_tool_calling(model_name: str, label: str):\n """Send a tool calling request and display the response."""\n response = client.inference.gateway.model.post(\n "v1/chat/completions",\n name=model_name,\n body={\n "messages": test_messages,\n "tools": test_tools,\n "tool_choice": "auto",\n "temperature": 0,\n "max_tokens": 256,\n },\n )\n\n print(f"{'=' * 60}")\n print(f" {label}")\n print(f"{'=' * 60}")\n print(json.dumps(response, indent=2))\n\n\ntest_tool_calling(MODEL_NAME, "BASE MODEL (before fine-tuning)")\ntest_tool_calling(ADAPTER_NAME, "FINE-TUNED MODEL (after LoRA)")\n" + "source_html": "test_messages = [\n {"role": "user", "content": "Calculate the factorial of 12 using math functions."},\n]\n\ntest_tools = [\n {\n "type": "function",\n "function": {\n "name": "math_factorial",\n "description": "Calculate the factorial of a given number.",\n "parameters": {\n "type": "object",\n "properties": {\n "number": {\n "type": "integer",\n "description": "The number for which factorial needs to be calculated.",\n }\n },\n "required": ["number"],\n },\n },\n }\n]\n\n\ndef test_tool_calling(model_name: str, label: str):\n """Send a tool calling request and display the response."""\n response = client.inference.gateway.provider.post(\n "v1/chat/completions",\n name=deployment_name,\n workspace=WORKSPACE,\n body={\n "model": model_name,\n "messages": test_messages,\n "tools": test_tools,\n "tool_choice": "auto",\n "temperature": 0,\n "max_tokens": 256,\n },\n )\n\n print(f"{'=' * 60}")\n print(f" {label}")\n print(f"{'=' * 60}")\n print(json.dumps(response, indent=2))\n\n\ntest_tool_calling(MODEL_NAME, "BASE MODEL (before fine-tuning)")\ntest_tool_calling(ADAPTER_NAME, "FINE-TUNED MODEL (after LoRA)")\n" }, { "type": "markdown", diff --git a/docs/fern/gated-nav.yml b/docs/fern/gated-nav.yml index e210ce8bdc..b9d6128dd9 100644 --- a/docs/fern/gated-nav.yml +++ b/docs/fern/gated-nav.yml @@ -145,8 +145,3 @@ contents: - page: Cluster Setup path: ../../troubleshooting/cluster-setup.mdx -- section: Fine-tune Models - contents: - # DPO is not yet a documented training type; the tutorial stays in the repo but out of nav. - - page: DPO Customization Job - path: ../../customizer/tutorials/dpo-customization-job.mdx diff --git a/docs/fern/scripts/README.md b/docs/fern/scripts/README.md index 75abc8bdde..ef43050b23 100644 --- a/docs/fern/scripts/README.md +++ b/docs/fern/scripts/README.md @@ -1,25 +1,21 @@ # Fern scripts +Run these from the **repository root**. For local development, use `uv run` — dependencies are resolved via the workspace `pyproject.toml`. + +`requirements.txt` in this directory lists the direct Python dependencies for CI or other `pip install -r` workflows. + ## `ipynb-to-fern-json.py` Converts Jupyter notebooks to the JSON/TS format consumed by `fern/components/NotebookViewer.tsx`. Pulled from [NVIDIA-NeMo/DataDesigner](https://github.com/NVIDIA-NeMo/DataDesigner/blob/main/fern/scripts/ipynb-to-fern-json.py). -### Setup - -```bash -python3 -m venv .venv -source .venv/bin/activate -pip install -r fern/scripts/requirements.txt -``` - ### Run ```bash -python fern/scripts/ipynb-to-fern-json.py \ +uv run python docs/fern/scripts/ipynb-to-fern-json.py \ docs/customizer/tutorials/sft-customization-job.ipynb \ - -o fern/components/notebooks/sft-customization-job.json + -o docs/fern/components/notebooks/sft-customization-job.json ``` Writes both `.json` (canonical data) and `.ts` (default-export wrapper @@ -37,3 +33,26 @@ After writing the `.ts` module, register it in `fern/components/NotebookViewer.t colabUrl="https://colab.research.google.com/github/NVIDIA-NeMo/nemo-platform/blob/main/docs/customizer/tutorials/sft-customization-job.ipynb" /> ``` + +## `ipynb-to-mdx.py` + +Converts Jupyter notebooks to **inline Fern MDX** using `nemo_nb` (`NotebookConverter`, +same engine as `nemo-nb to-sphinx-md`). Post-processes for Fern frontmatter, a Google +Colab banner, and canonical `/documentation/...` internal links. + +### Run MDX conversion + +```bash +uv run python docs/fern/scripts/ipynb-to-mdx.py --all-customizer-tutorials +``` + +Or a single notebook: + +```bash +uv run python docs/fern/scripts/ipynb-to-mdx.py \ + docs/customizer/tutorials/sft-customization-job.ipynb \ + -o docs/customizer/tutorials/sft-customization-job.mdx \ + --title "Full SFT Customization" +``` + +Re-run whenever the source `.ipynb` changes. diff --git a/docs/fern/scripts/ipynb-to-fern-json.py b/docs/fern/scripts/ipynb-to-fern-json.py index 9ca81d041f..5ce1128372 100644 --- a/docs/fern/scripts/ipynb-to-fern-json.py +++ b/docs/fern/scripts/ipynb-to-fern-json.py @@ -32,8 +32,19 @@ import json import re import sys +from datetime import datetime from pathlib import Path +_CURRENT_YEAR = datetime.now().year +_TS_FILE_HEADER = ( + "/**\n" + f" * SPDX-FileCopyrightText: Copyright (c) 2025-{_CURRENT_YEAR} NVIDIA CORPORATION & AFFILIATES. All rights reserved.\n" + " * SPDX-License-Identifier: Apache-2.0\n" + " *\n" + " * Auto-generated by ipynb-to-fern-json.py - do not edit manually.\n" + " */\n" +) + from markdown_it import MarkdownIt from PIL import Image from pygments import highlight @@ -302,7 +313,7 @@ def write_ts_export(data: dict, ts_path: Path) -> None: """Write a .ts file that exports the notebook data inline (MDX imports the .ts, not the .json).""" cells_json = json.dumps(data["cells"], indent=2, ensure_ascii=False) ts_path.write_text( - f"/** Auto-generated by ipynb-to-fern-json.py - do not edit */\nexport default {{ cells: {cells_json} }};\n", + f"{_TS_FILE_HEADER}export default {{ cells: {cells_json} }};\n", encoding="utf-8", ) diff --git a/docs/fern/scripts/ipynb-to-mdx.py b/docs/fern/scripts/ipynb-to-mdx.py new file mode 100644 index 0000000000..f8841e6209 --- /dev/null +++ b/docs/fern/scripts/ipynb-to-mdx.py @@ -0,0 +1,178 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Convert Jupyter notebooks to inline Fern MDX using nemo-nb. + +Wraps ``nemo_nb.converter.NotebookConverter`` (``nemo-nb to-sphinx-md``) and +post-processes the output for Fern: + - Fern frontmatter (title, description) + - Google Colab link instead of the nemo-nb download anchor + - Relative doc links rewritten to canonical ``/documentation/...`` URLs + +Usage: + python ipynb-to-mdx.py input.ipynb -o output.mdx --title "Page Title" + python ipynb-to-mdx.py --all-customizer-tutorials +""" + +from __future__ import annotations + +import re +import sys +from pathlib import Path + +from nemo_nb.converter import NotebookConverter + +COLAB_REPO = "https://colab.research.google.com/github/NVIDIA-NeMo/nemo-platform/blob/main" + +DOWNLOAD_LINK_RE = re.compile( + r'Download this tutorial as a Jupyter notebook\s*', + re.IGNORECASE, +) + +_LINK_REWRITES: list[tuple[re.Pattern[str], str]] = [ + ( + re.compile(r"\]\(\.\./\.\./get-started/quickstart\.md?\)"), + "](/documentation/get-started)", + ), + ( + re.compile(r"\]\(\.\./\.\./get-started/concepts/manage-secrets\.md?\)"), + "](/documentation/get-started/core-concepts/manage-secrets)", + ), + ( + re.compile(r"\]\(\.\./manage-customization-jobs/hyperparameters\.md?\)"), + "](/documentation/customizer-reference/manage-customization-jobs/training-configuration)", + ), + ( + re.compile(r"\]\(\.\./manage-customization-jobs/get-job-status\.md?\)"), + "](/documentation/customizer-reference/manage-customization-jobs/get-job-status)", + ), + ( + re.compile(r"\]\(\.\./\.\./evaluator/index(?:\.md)?\)"), + "](/documentation/evaluate-models)", + ), + ( + re.compile(r"\]\(\./distillation-customization-job(?:\.ipynb)?\)"), + "](/documentation/customizer-reference/tutorials/distillation-customization-job)", + ), + ( + re.compile(r"\]\(\./embedding-customization-job(?:\.ipynb)?\)"), + "](/documentation/customizer-reference/tutorials/embedding-customization-job)", + ), + ( + re.compile(r"\]\(\./lora-customization-job(?:\.ipynb)?\)"), + "](/documentation/customizer-reference/tutorials/lora-customization-job)", + ), + ( + re.compile(r"\]\(\./optimize-throughput(?:\.ipynb)?\)"), + "](/documentation/customizer-reference/tutorials/optimize-throughput)", + ), + ( + re.compile(r"\]\(\./sft-customization-job(?:\.ipynb)?\)"), + "](/documentation/customizer-reference/tutorials/sft-customization-job)", + ), + ( + re.compile(r"\]\(fine-tune-metrics\)"), + "](/documentation/customizer-reference/tutorials/metrics)", + ), + ( + re.compile(r"\]\(nemo-ms-about-concepts-customization\)"), + "](/documentation/customizer-reference/customization-concepts#nemo-ms-about-concepts-customization)", + ), +] + +CUSTOMIZER_TUTORIALS: list[tuple[str, str]] = [ + ("sft-customization-job.ipynb", "Full SFT Customization"), + ("lora-customization-job.ipynb", "LoRA Model Customization"), + ("distillation-customization-job.ipynb", "Knowledge Distillation Customization"), + ("embedding-customization-job.ipynb", "Embedding Model Customization"), + ("optimize-throughput.ipynb", "Optimize for Tokens/GPU Throughput"), +] + + +def rewrite_links(text: str) -> str: + for pattern, replacement in _LINK_REWRITES: + text = pattern.sub(replacement, text) + return text + + +def repo_relative_path(path: Path) -> str: + repo_root = Path(__file__).resolve().parents[3] + return path.resolve().relative_to(repo_root).as_posix() + + +def colab_link_for(ipynb_path: Path) -> str: + return f"[Run in Google Colab]({COLAB_REPO}/{repo_relative_path(ipynb_path)})" + + +def convert_notebook_to_mdx(ipynb_path: Path, *, title: str) -> str: + body = NotebookConverter().convert(ipynb_path) + body = DOWNLOAD_LINK_RE.sub("", body).lstrip("\n") + body = rewrite_links(body) + + return ( + "---\n" + f'title: "{title}"\n' + 'description: ""\n' + "---\n" + "\n" + f"{colab_link_for(ipynb_path)}\n" + "\n" + f"{body.rstrip()}\n" + ) + + +def convert_all_customizer_tutorials(tutorials_dir: Path) -> int: + rc = 0 + for notebook_name, title in CUSTOMIZER_TUTORIALS: + ipynb = tutorials_dir / notebook_name + mdx = tutorials_dir / notebook_name.replace(".ipynb", ".mdx") + if not ipynb.exists(): + print(f"Error: {ipynb} not found", file=sys.stderr) + rc = 1 + continue + mdx.write_text(convert_notebook_to_mdx(ipynb, title=title), encoding="utf-8") + print(f"Wrote {mdx}") + return rc + + +def main() -> int: + args = sys.argv[1:] + if not args or "-h" in args or "--help" in args: + print(__doc__) + return 0 + + if "--all-customizer-tutorials" in args: + repo_root = Path(__file__).resolve().parents[3] + tutorials_dir = repo_root / "docs" / "customizer" / "tutorials" + return convert_all_customizer_tutorials(tutorials_dir) + + input_path = Path(args[0]) + output_path: Path | None = None + title: str | None = None + + if "-o" in args: + idx = args.index("-o") + if idx + 1 < len(args): + output_path = Path(args[idx + 1]) + if "--title" in args: + idx = args.index("--title") + if idx + 1 < len(args): + title = args[idx + 1] + + if not input_path.exists(): + print(f"Error: {input_path} not found", file=sys.stderr) + return 1 + if output_path is None: + output_path = input_path.with_suffix(".mdx") + if not title: + print("Error: --title is required", file=sys.stderr) + return 1 + + output_path.write_text(convert_notebook_to_mdx(input_path, title=title), encoding="utf-8") + print(f"Wrote {output_path}") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/docs/get-started/concepts/filtering.mdx b/docs/get-started/concepts/filtering.mdx index 2ac7bd2f27..304b428b34 100644 --- a/docs/get-started/concepts/filtering.mdx +++ b/docs/get-started/concepts/filtering.mdx @@ -180,5 +180,8 @@ metrics = client.evaluation.metrics.list( ### Filtering by status ```python -jobs = client.customization.jobs.list(filter={"status": "completed"}) +jobs = client.jobs.list( + workspace="default", + filter={"source": "automodel", "status": "completed"}, +) ``` diff --git a/docs/troubleshooting/customizer.mdx b/docs/troubleshooting/customizer.mdx index 3b90b8b408..c5a994e343 100644 --- a/docs/troubleshooting/customizer.mdx +++ b/docs/troubleshooting/customizer.mdx @@ -5,7 +5,7 @@ description: "" **Job fails during model download:** - Verify the HuggingFace token secret is configured correctly - Accept the model's license on the [HuggingFace model page](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) -- Check job status: `client.customization.jobs.retrieve(name=job.name, workspace="default")` +- Check job status: `client.jobs.get_status(name=job.name, workspace="default")` **Job fails with disk full or 500 error when retrieving logs:** - The platform's shared persistent volume is likely full. Customization jobs require significant disk space: ~3× model size for full SFT, ~1.5× for LoRA. If you are also deploying the model from a base checkpoint fileset, plan for ~2.5× model size overall. diff --git a/packages/nmp_testing/src/nmp/testing/e2e/customizer.py b/packages/nmp_testing/src/nmp/testing/e2e/customizer.py index c0a9180b5e..ff2f837d59 100644 --- a/packages/nmp_testing/src/nmp/testing/e2e/customizer.py +++ b/packages/nmp_testing/src/nmp/testing/e2e/customizer.py @@ -218,7 +218,7 @@ def get_job_failure_details(sdk: NeMoPlatform, job_name: str, workspace: str) -> def _log_training_progress(status) -> None: """Extract and log training progress from the job status steps structure.""" for job_step in status.steps or []: - if job_step.name == "customization-training-job": + if job_step.name == "training": for task in job_step.tasks or []: task_details = task.status_details or {} step = task_details.get("step") diff --git a/plugins/nemo-automodel/src/nemo_automodel_plugin/transform.py b/plugins/nemo-automodel/src/nemo_automodel_plugin/transform.py index d9bf668f05..a3a3c70647 100644 --- a/plugins/nemo-automodel/src/nemo_automodel_plugin/transform.py +++ b/plugins/nemo-automodel/src/nemo_automodel_plugin/transform.py @@ -41,11 +41,6 @@ async def transform_input_to_output( await check_dataset_access(sdk, input_spec.dataset.validation, workspace) is_embedding = bool(model_entity.spec and getattr(model_entity.spec, "is_embedding_model", False)) - if is_embedding: - raise ValueError( - "Embedding-model SFT is not supported in Automodel v1. " - "Use a causal LM checkpoint or wait for a future release." - ) output_type = _infer_output_type(input_spec, is_embedding) diff --git a/services/automodel/src/nmp/automodel/tasks/training/backends/checkpoints.py b/services/automodel/src/nmp/automodel/tasks/training/backends/checkpoints.py index f43220fe2f..770fd73aae 100644 --- a/services/automodel/src/nmp/automodel/tasks/training/backends/checkpoints.py +++ b/services/automodel/src/nmp/automodel/tasks/training/backends/checkpoints.py @@ -365,7 +365,7 @@ def export_onnx( tokenizer_path=tokenizer_path, pooling="avg", normalize=True, - opset=17, + opset=18, export_dtype="fp16", verify=True, ) diff --git a/services/core/models/src/nmp/core/models/api/v2/models.py b/services/core/models/src/nmp/core/models/api/v2/models.py index aa84d33ffd..4a8d80fdef 100644 --- a/services/core/models/src/nmp/core/models/api/v2/models.py +++ b/services/core/models/src/nmp/core/models/api/v2/models.py @@ -269,6 +269,10 @@ async def start_update_model_spec_job(model_entity: ModelEntity): name="model-spec-analysis", executor=CPUExecutionProviderSpec( provider="cpu", + # Profile ``gpu`` selects the Docker CPU executor (see jobs.executors + # in platform config). Avoid ``default``, which is translated to the + # host subprocess backend when subprocess/default is registered. + profile="gpu", container=ContainerSpec( image=get_qualified_image("nmp-automodel-tasks"), entrypoint=["/opt/venv/bin/python"],