NANInithin
Claude Opus 4.8
feat(training): LoRA fine-tune pipeline for Nemotron 3 Nano 4B on Modal
7bb583b Download training/train_modal.py from build-small-hackathon/CityQuest-AI: direct link, hf CLI and curl.
- Browser
- Download file 27.9 kB
-
https://huggingface.co/spaces/build-small-hackathon/CityQuest-AI/resolve/main/training/train_modal.py
- Command line
-
hf download hf://spaces/build-small-hackathon/CityQuest-AI/training/train_modal.py
-
curl -L -o train_modal.py https://huggingface.co/spaces/build-small-hackathon/CityQuest-AI/resolve/main/training/train_modal.py
27.9 kB
| """Fine-tune NVIDIA Nemotron 3 Nano 4B on CityQuest data (LoRA) on Modal, then | |
| merge, convert to GGUF Q4_K_M and upload to Hugging Face. | |
| The model (`nemotron_h`, a hybrid Mamba-Transformer) is loaded **directly from | |
| Hugging Face with plain `transformers` + PEFT** — no Unsloth (its model-type | |
| detector doesn't recognise this brand-new architecture). transformers 5.x ships a | |
| native NemotronH implementation with a pure-PyTorch fallback, so the CUDA Mamba | |
| kernels (mamba-ssm) are optional; training just runs a bit slower without them. | |
| Pipeline (one A100-80GB): | |
| 1. Load base (bf16) from HF, attach LoRA to all linear layers (router/lm_head excluded). | |
| 2. SFT on training/data/train.jsonl with response-only loss masking; eval on val.jsonl. | |
| 3. Merge LoRA → 16-bit safetensors (+ tokenizer & chat template). | |
| 4. Convert merged model → GGUF Q4_K_M via llama.cpp. | |
| 5. Upload the .gguf to HF_REPO so generator.py can pull it. | |
| Run (after `modal token new` + the HF secret — see training/README.md): | |
| modal run training/train_modal.py::main | |
| """ | |
| from __future__ import annotations | |
| import os | |
| from pathlib import Path | |
| import modal | |
| # ── Configuration ───────────────────────────────────────────────────────────── | |
| BASE_MODEL = "nvidia/NVIDIA-Nemotron-3-Nano-4B-BF16" # official full-precision base on HF | |
| HF_REPO = "NANI-Nithin/CityQuest-Nemotron-3-Nano-4B-GGUF" | |
| GGUF_OUTFILE = "CityQuest-Nemotron-3-Nano-4B-Q4_K_M.gguf" | |
| # llama.cpp commit whose convert_hf_to_gguf handles the DENSE nemotron_h 4B | |
| # (earlier master misclassified it as MoE → KeyError 'moe_intermediate_size'). | |
| LLAMA_CPP_COMMIT = "e36a602ba38a26206c749ba4fb5dcf481bfd92db" | |
| # Stock baseline GGUF (matches generator.py's current constants) for comparison. | |
| BASELINE_REPO = "nvidia/NVIDIA-Nemotron-3-Nano-4B-GGUF" | |
| BASELINE_FILE = "NVIDIA-Nemotron3-Nano-4B-Q4_K_M.gguf" | |
| MAX_SEQ_LEN = 8192 # matches serve-time n_ctx; covers our longest sample (~5.4k) | |
| NUM_EPOCHS = 3 | |
| LEARNING_RATE = 2e-4 | |
| LORA_R = 16 | |
| LORA_ALPHA = 16 | |
| BATCH_SIZE = 1 | |
| GRAD_ACCUM = 16 # effective batch 16 | |
| ASSISTANT_HEADER = "<|im_start|>assistant\n" # from the model's chat template | |
| LOCAL_DATA = Path(__file__).resolve().parent / "data" | |
| REMOTE_DATA = "/root/data" | |
| LOCAL_SCHEMA = Path(__file__).resolve().parent.parent / "app" / "schemas" / "game_schema.json" | |
| REMOTE_SCHEMA = "/root/game_schema.json" | |
| # ── Training image ───────────────────────────────────────────────────────── | |
| # CUDA-devel base (provides nvcc) so we can build the Mamba CUDA kernels. Without | |
| # mamba-ssm/causal-conv1d the native nemotron_h falls back to a naive O(L^2) Mamba | |
| # path that OOMs at our sequence lengths — the fused kernels are required here. | |
| image = ( | |
| modal.Image.from_registry("nvidia/cuda:12.4.1-devel-ubuntu22.04", add_python="3.11") | |
| .apt_install("git", "build-essential", "cmake", "curl", "libcurl4-openssl-dev") | |
| .env({ | |
| "TORCH_CUDA_ARCH_LIST": "8.0", "MAX_JOBS": "4", "CC": "gcc", "CXX": "g++", | |
| "CAUSAL_CONV1D_FORCE_BUILD": "TRUE", "MAMBA_FORCE_BUILD": "TRUE", | |
| }) | |
| # Single controlled install pass. A constraints file pins torch==2.6.0 so NOTHING | |
| # (transformers/peft/accelerate) can swap it for the PyPI cu130 build mid-install — | |
| # that version drift is what broke the kernel ABI (undefined c10::cuda symbol). | |
| # The Mamba kernels are force-compiled FROM SOURCE against this exact torch, then | |
| # imported AT BUILD TIME so any ABI mismatch fails the build (CPU builder, ~1 min, | |
| # no GPU burned) with the torch version printed. | |
| .run_commands( | |
| "pip install --upgrade pip", | |
| "pip install torch==2.6.0 --index-url https://download.pytorch.org/whl/cu124", | |
| "printf 'torch==2.6.0\\n' > /root/constraints.txt", | |
| "pip install -c /root/constraints.txt ninja packaging wheel setuptools " | |
| "'transformers>=5.4.0' 'peft>=0.13.0' 'accelerate>=1.0.0' 'datasets>=3.0.0' " | |
| "huggingface_hub einops sentencepiece protobuf gguf 'numpy<2.0'", | |
| "python -c \"import torch; print('[build] torch', torch.__version__, torch.version.cuda)\"", | |
| "pip install -c /root/constraints.txt --no-build-isolation causal-conv1d", | |
| # Pin 2.3.1: newer 2.3.2.post1 drags in tilelang/triton>=3.5 which conflict | |
| # with torch 2.6's triton 3.2 (pip would backtrack to 2.3.1 anyway). | |
| "pip install -c /root/constraints.txt --no-build-isolation mamba-ssm==2.3.1", | |
| # transformers' lazy_load_kernel prefers the HF `kernels` hub package (no working | |
| # mamba build → silent fallback to the OOM naive path). Remove it so transformers | |
| # imports the mamba_ssm/causal_conv1d we just compiled. | |
| "pip uninstall -y kernels || true", | |
| # Build-time ABI gate: import the compiled CUDA C-extension only (this is what | |
| # had the undefined-symbol error). It validates the torch ABI without triggering | |
| # Triton autotune, which needs a GPU the CPU image-builder doesn't have. | |
| "python -c \"import torch, selective_scan_cuda; " | |
| "print('[build] CUDA ext ABI OK on torch', torch.__version__)\"", | |
| ) | |
| .add_local_dir(str(LOCAL_DATA), remote_path=REMOTE_DATA) | |
| ) | |
| # Eval image: CUDA *runtime* base so the cu124 llama-cpp-python wheel finds | |
| # libcudart.so.12 (debian_slim lacks the CUDA runtime libs). | |
| eval_image = ( | |
| modal.Image.from_registry("nvidia/cuda:12.4.1-runtime-ubuntu22.04", add_python="3.11") | |
| .pip_install("huggingface_hub", "hf_transfer", "jsonschema") | |
| .pip_install( | |
| "llama-cpp-python", | |
| extra_index_url="https://abetlen.github.io/llama-cpp-python/whl/cu124", | |
| ) | |
| .env({"HF_HUB_ENABLE_HF_TRANSFER": "1"}) | |
| .add_local_dir(str(LOCAL_DATA), remote_path=REMOTE_DATA) | |
| .add_local_file(str(LOCAL_SCHEMA), remote_path=REMOTE_SCHEMA) | |
| ) | |
| app = modal.App("cityquest-nemotron-finetune", image=image) | |
| hf_cache = modal.Volume.from_name("cityquest-hf-cache", create_if_missing=True) | |
| outputs_vol = modal.Volume.from_name("cityquest-outputs", create_if_missing=True) | |
| def _find_subsequence(seq: list[int], sub: list[int]) -> int: | |
| """Return the start index of the last occurrence of ``sub`` in ``seq``, else -1.""" | |
| if not sub: | |
| return -1 | |
| for start in range(len(seq) - len(sub), -1, -1): | |
| if seq[start:start + len(sub)] == sub: | |
| return start | |
| return -1 | |
| def train(): | |
| import json | |
| import os | |
| import torch | |
| import torch.nn as nn | |
| from datasets import load_dataset | |
| from peft import LoraConfig, get_peft_model | |
| from transformers import ( | |
| AutoModelForCausalLM, | |
| AutoTokenizer, | |
| DataCollatorForSeq2Seq, | |
| Trainer, | |
| TrainerCallback, | |
| TrainingArguments, | |
| ) | |
| from transformers.trainer_utils import get_last_checkpoint | |
| merged_dir = "/root/outputs/merged_16bit" | |
| gguf_dir = "/root/outputs/gguf" | |
| # ── 0. Mamba-kernel diagnostic + hard gate ───────────────────────────── | |
| # The native nemotron_h falls back to a naive O(L^2) Mamba path (OOMs at our | |
| # sequence lengths) unless mamba-ssm/causal-conv1d import cleanly. Surface the | |
| # real reason and fail fast (seconds) instead of OOMing after model load. | |
| print(f"[diag] torch={torch.__version__} cuda_avail={torch.cuda.is_available()} " | |
| f"torch.cuda={torch.version.cuda}") | |
| try: | |
| import kernels # noqa: F401 | |
| print("[diag] WARNING: `kernels` package still installed (may hijack kernel loading)") | |
| except ImportError: | |
| print("[diag] `kernels` package absent (good — will import compiled mamba_ssm)") | |
| from transformers.utils.import_utils import ( | |
| is_causal_conv1d_available, is_mamba_2_ssm_available, | |
| ) | |
| print(f"[diag] is_mamba_2_ssm_available={is_mamba_2_ssm_available()} " | |
| f"is_causal_conv1d_available={is_causal_conv1d_available()}") | |
| _kernel_err = None | |
| try: | |
| import mamba_ssm | |
| from mamba_ssm.ops.triton.ssd_combined import mamba_chunk_scan_combined # noqa: F401 | |
| from mamba_ssm.ops.triton.selective_state_update import selective_state_update # noqa: F401 | |
| import causal_conv1d # noqa: F401 | |
| from causal_conv1d import causal_conv1d_fn # noqa: F401 | |
| print(f"[diag] mamba_ssm {getattr(mamba_ssm, '__version__', '?')} + causal_conv1d import OK") | |
| except Exception as e: # noqa: BLE001 | |
| import traceback | |
| _kernel_err = traceback.format_exc() | |
| print(f"[diag] KERNEL IMPORT FAILED:\n{_kernel_err}") | |
| if _kernel_err is not None: | |
| raise RuntimeError( | |
| "Mamba CUDA kernels not importable — training would fall back to the " | |
| "OOM-prone naive path. See [diag] traceback above for the root cause." | |
| ) | |
| # ── 1. Load base + LoRA (native transformers, no Unsloth) ────────────── | |
| # IMPORTANT: trust_remote_code=False → use transformers' *native* nemotron_h | |
| # implementation, which falls back to a pure-PyTorch path when the mamba-ssm / | |
| # causal-conv1d CUDA kernels are absent (it only warns). NVIDIA's *remote* code | |
| # hard-raises ImportError without those kernels. | |
| print(f"[train] loading base from HF: {BASE_MODEL}") | |
| tokenizer = AutoTokenizer.from_pretrained(BASE_MODEL) | |
| if tokenizer.pad_token is None: | |
| tokenizer.pad_token = tokenizer.eos_token # this model ships no pad token | |
| model = AutoModelForCausalLM.from_pretrained( | |
| BASE_MODEL, | |
| dtype=torch.bfloat16, | |
| low_cpu_mem_usage=True, | |
| attn_implementation="sdpa", # memory-efficient attention for the transformer layers | |
| ) | |
| model.config.pad_token_id = tokenizer.pad_token_id | |
| model.config.use_cache = False | |
| model.gradient_checkpointing_enable(gradient_checkpointing_kwargs={"use_reentrant": False}) | |
| model.enable_input_require_grads() | |
| # Discover LoRA targets: every nn.Linear leaf name except the LM head and any | |
| # MoE router (router weights shouldn't be adapted). Covers attention + MLP + | |
| # Mamba projections regardless of the exact arch naming. | |
| target_modules = set() | |
| for name, module in model.named_modules(): | |
| if isinstance(module, nn.Linear): | |
| leaf = name.split(".")[-1] | |
| if leaf in ("lm_head",) or "router" in name.lower(): | |
| continue | |
| target_modules.add(leaf) | |
| target_modules = sorted(target_modules) | |
| print(f"[train] LoRA target modules: {target_modules}") | |
| model = get_peft_model(model, LoraConfig( | |
| r=LORA_R, lora_alpha=LORA_ALPHA, lora_dropout=0.0, bias="none", | |
| task_type="CAUSAL_LM", target_modules=target_modules, | |
| )) | |
| model.print_trainable_parameters() | |
| # ── 2. Tokenize with response-only label masking ─────────────────────── | |
| header_ids = tokenizer(ASSISTANT_HEADER, add_special_tokens=False)["input_ids"] | |
| def tokenize_and_mask(row): | |
| # enable_thinking=False keeps non-reasoning/JSON-only mode (matches the | |
| # serve-time "output only valid JSON" system prompt). | |
| try: | |
| text = tokenizer.apply_chat_template( | |
| row["messages"], tokenize=False, add_generation_prompt=False, | |
| enable_thinking=False, | |
| ) | |
| except TypeError: | |
| text = tokenizer.apply_chat_template( | |
| row["messages"], tokenize=False, add_generation_prompt=False, | |
| ) | |
| enc = tokenizer(text, truncation=True, max_length=MAX_SEQ_LEN, add_special_tokens=False) | |
| input_ids = enc["input_ids"] | |
| labels = list(input_ids) | |
| cut = _find_subsequence(input_ids, header_ids) | |
| if cut != -1: | |
| for i in range(cut + len(header_ids)): | |
| labels[i] = -100 # mask system + user; train only on the JSON answer | |
| enc["labels"] = labels | |
| return enc | |
| train_ds = load_dataset("json", data_files=f"{REMOTE_DATA}/train.jsonl", split="train") | |
| val_ds = load_dataset("json", data_files=f"{REMOTE_DATA}/val.jsonl", split="train") | |
| train_ds = train_ds.map(tokenize_and_mask, remove_columns=train_ds.column_names) | |
| val_ds = val_ds.map(tokenize_and_mask, remove_columns=val_ds.column_names) | |
| print(f"[train] train={len(train_ds)} val={len(val_ds)}") | |
| collator = DataCollatorForSeq2Seq(tokenizer, label_pad_token_id=-100, padding="longest") | |
| # Checkpoint to the volume every few steps so a Modal preemption (spot GPU | |
| # reclaim) resumes from the last step instead of restarting at 0. A callback | |
| # commits the volume after each save so checkpoints survive an abrupt restart. | |
| ckpt_dir = "/root/outputs/checkpoints" | |
| class _CommitCheckpoint(TrainerCallback): | |
| def on_save(self, args, state, control, **kwargs): | |
| outputs_vol.commit() | |
| print(f"[train] checkpoint committed @ step {state.global_step}") | |
| trainer = Trainer( | |
| model=model, | |
| args=TrainingArguments( | |
| output_dir=ckpt_dir, | |
| per_device_train_batch_size=BATCH_SIZE, | |
| gradient_accumulation_steps=GRAD_ACCUM, | |
| warmup_ratio=0.05, | |
| num_train_epochs=NUM_EPOCHS, | |
| learning_rate=LEARNING_RATE, | |
| lr_scheduler_type="cosine", | |
| bf16=True, | |
| logging_steps=10, | |
| eval_strategy="epoch", | |
| save_strategy="steps", | |
| save_steps=20, | |
| save_total_limit=1, | |
| optim="adamw_torch", | |
| seed=13, | |
| report_to="none", | |
| gradient_checkpointing=True, | |
| gradient_checkpointing_kwargs={"use_reentrant": False}, | |
| ), | |
| train_dataset=train_ds, | |
| eval_dataset=val_ds, | |
| data_collator=collator, | |
| callbacks=[_CommitCheckpoint()], | |
| ) | |
| # ── 3. Train (resume from last checkpoint if a preemption left one) ───── | |
| last_ckpt = get_last_checkpoint(ckpt_dir) if os.path.isdir(ckpt_dir) else None | |
| if last_ckpt: | |
| print(f"[train] resuming from checkpoint: {last_ckpt}") | |
| stats = trainer.train(resume_from_checkpoint=last_ckpt) | |
| print(f"[train] done: {stats.metrics}") | |
| # ── 3b. Save the LoRA adapter first + commit (cheap recovery point) ───── | |
| # Persist the trained adapter before the merge/GGUF steps so a later failure | |
| # never costs us the ~40 min of training again. | |
| adapter_dir = "/root/outputs/lora_adapter" | |
| model.save_pretrained(adapter_dir) | |
| tokenizer.save_pretrained(adapter_dir) | |
| outputs_vol.commit() | |
| print(f"[train] saved LoRA adapter → {adapter_dir} (committed)") | |
| # ── 4. Merge LoRA → 16-bit, then convert to GGUF + upload ────────────── | |
| print("[train] merging LoRA → 16-bit safetensors") | |
| merged = model.merge_and_unload() | |
| _save_convert_upload(merged, tokenizer, merged_dir, gguf_dir) | |
| def finalize(): | |
| """Resume from the committed LoRA adapter: re-merge → GGUF → upload. | |
| Lets us recover a completed training run without retraining (e.g. when only the | |
| GGUF/upload tail failed). Needs a GPU because instantiating nemotron_h imports | |
| the Triton Mamba kernels. | |
| """ | |
| import torch | |
| from peft import PeftModel | |
| from transformers import AutoModelForCausalLM, AutoTokenizer | |
| adapter_dir = "/root/outputs/lora_adapter" | |
| merged_dir = "/root/outputs/merged_16bit" | |
| gguf_dir = "/root/outputs/gguf" | |
| print(f"[finalize] loading base {BASE_MODEL} + adapter {adapter_dir}") | |
| tokenizer = AutoTokenizer.from_pretrained(adapter_dir) | |
| base = AutoModelForCausalLM.from_pretrained( | |
| BASE_MODEL, dtype=torch.bfloat16, low_cpu_mem_usage=True, attn_implementation="sdpa", | |
| ) | |
| model = PeftModel.from_pretrained(base, adapter_dir) | |
| merged = model.merge_and_unload() | |
| _save_convert_upload(merged, tokenizer, merged_dir, gguf_dir) | |
| def convert_from_merged(): | |
| """Convert an already-committed merged model → GGUF → upload. No GPU, no peft. | |
| The fast/cheap resume path when train()'s GGUF/upload tail fails but the merged | |
| model was committed first (it is, before the GGUF step). convert_hf_to_gguf reads | |
| config + safetensors directly — no model instantiation, so no Mamba kernels/GPU. | |
| """ | |
| import os | |
| from huggingface_hub import HfApi | |
| merged_dir = "/root/outputs/merged_16bit" | |
| gguf_dir = "/root/outputs/gguf" | |
| _prepare_config_for_conversion(merged_dir) | |
| gguf_path = _convert_with_llama_cpp(merged_dir, gguf_dir) | |
| outputs_vol.commit() | |
| api = HfApi(token=os.environ["HF_TOKEN"]) | |
| api.create_repo(HF_REPO, repo_type="model", exist_ok=True) | |
| api.upload_file(path_or_fileobj=gguf_path, path_in_repo=GGUF_OUTFILE, | |
| repo_id=HF_REPO, repo_type="model") | |
| api.upload_file(path_or_fileobj=_model_card().encode("utf-8"), path_in_repo="README.md", | |
| repo_id=HF_REPO, repo_type="model") | |
| print(f"[convert] uploaded {GGUF_OUTFILE} → https://huggingface.co/{HF_REPO}") | |
| def _prepare_config_for_conversion(model_dir: str) -> None: | |
| """Overwrite the merged config.json with the *original base* config — verbatim, | |
| including ``auto_map`` — so llama.cpp's converter reads the right hparams. | |
| The converter's ``load_hparams`` does ``AutoConfig.from_pretrained( | |
| trust_remote_code=False).to_dict()``. For nemotron_h that returns NemotronHConfig | |
| with its *class-default* MoE attributes (``num_experts_per_tok``), which makes the | |
| converter think the dense 4B is MoE → ``KeyError: moe_intermediate_size``. | |
| Keeping ``auto_map`` makes that AutoConfig call refuse the custom code and fall | |
| back to the *raw* config.json — the base config, which has ``num_hidden_layers`` | |
| and ``intermediate_size`` but NO expert keys → converter takes the dense path. | |
| (transformers 5.12's own re-save dropped ``num_hidden_layers``, so we must use the | |
| base config, not the merged one.) | |
| """ | |
| import json | |
| from huggingface_hub import hf_hub_download | |
| base_cfg = hf_hub_download(BASE_MODEL, "config.json") | |
| data = json.loads(Path(base_cfg).read_text()) # keep auto_map intact on purpose | |
| (Path(model_dir) / "config.json").write_text(json.dumps(data, indent=2)) | |
| print(f"[finalize] wrote verbatim base config.json (auto_map kept → raw-JSON fallback) → {model_dir}") | |
| def _save_convert_upload(merged, tokenizer, merged_dir: str, gguf_dir: str) -> str: | |
| """Normalize config, save merged model, convert to GGUF Q4_K_M, upload to HF. | |
| Shared by ``train`` (in-memory merged model) and ``finalize`` (re-merged from | |
| the committed LoRA adapter). | |
| """ | |
| import os | |
| from huggingface_hub import HfApi | |
| # Nemotron ships generation_config with top_p set but do_sample=False, which | |
| # transformers' strict save-validation rejects. Normalize to clean greedy. | |
| gc = merged.generation_config | |
| gc.do_sample = False | |
| for _attr in ("top_p", "top_k", "typical_p", "temperature"): | |
| if getattr(gc, _attr, None) is not None: | |
| setattr(gc, _attr, None) | |
| merged.save_pretrained(merged_dir, safe_serialization=True) | |
| tokenizer.save_pretrained(merged_dir) | |
| _prepare_config_for_conversion(merged_dir) | |
| outputs_vol.commit() # persist merged model before the (fragile) GGUF step | |
| gguf_path = _convert_with_llama_cpp(merged_dir, gguf_dir) | |
| print(f"[finalize] GGUF ready: {gguf_path}") | |
| outputs_vol.commit() | |
| api = HfApi(token=os.environ["HF_TOKEN"]) | |
| api.create_repo(HF_REPO, repo_type="model", exist_ok=True) | |
| api.upload_file(path_or_fileobj=gguf_path, path_in_repo=GGUF_OUTFILE, | |
| repo_id=HF_REPO, repo_type="model") | |
| api.upload_file(path_or_fileobj=_model_card().encode("utf-8"), path_in_repo="README.md", | |
| repo_id=HF_REPO, repo_type="model") | |
| print(f"[finalize] uploaded {GGUF_OUTFILE} → https://huggingface.co/{HF_REPO}") | |
| return gguf_path | |
| def _convert_with_llama_cpp(merged_dir: str, gguf_dir: str) -> str: | |
| """Clone llama.cpp, convert merged HF model to f16 GGUF, quantize Q4_K_M. | |
| Uses latest llama.cpp, which supports the nemotron_h architecture (NVIDIA and | |
| others publish Nemotron-3 GGUFs built with it). | |
| """ | |
| import shutil | |
| import subprocess | |
| Path(gguf_dir).mkdir(parents=True, exist_ok=True) | |
| # Commit-specific path so a stale clone in a reused (warm) Modal container can't | |
| # be mistaken for the pinned version — that bug skipped the checkout and used an | |
| # old converter that misclassified the dense 4B as MoE. | |
| llama_dir = f"/root/llama.cpp-{LLAMA_CPP_COMMIT[:8]}" | |
| quantize_bin = f"{llama_dir}/build/bin/llama-quantize" | |
| convert_py = f"{llama_dir}/convert_hf_to_gguf.py" | |
| if not Path(quantize_bin).exists(): | |
| if Path(llama_dir).exists(): | |
| shutil.rmtree(llama_dir) | |
| subprocess.run(["git", "clone", "https://github.com/ggml-org/llama.cpp", llama_dir], check=True) | |
| # Pin to a commit whose converter has the nemotron_h dense/MoE split. | |
| subprocess.run(["git", "-C", llama_dir, "checkout", LLAMA_CPP_COMMIT], check=True) | |
| # NOTE: deliberately NOT `pip install -r requirements.txt` — it downgrades our | |
| # transformers below nemotron_h support. The convert deps are already in the image. | |
| subprocess.run(["cmake", "-S", llama_dir, "-B", f"{llama_dir}/build", "-DLLAMA_CURL=OFF"], check=True) | |
| subprocess.run(["cmake", "--build", f"{llama_dir}/build", "--target", "llama-quantize", "-j"], check=True) | |
| # Force the DENSE path (idempotent — run every call, NOT just on fresh clone, since | |
| # a warm container may already have the built dir and skip the block above). | |
| # transformers' NemotronHConfig always carries a default num_experts_per_tok=2, so | |
| # the converter's `if has_moe_params` always trips and treats our dense 4B as MoE | |
| # (→ KeyError moe_intermediate_size, and a broken empty-FFN GGUF). The 4B has no | |
| # experts, so disabling that branch is correct. | |
| subprocess.run( | |
| ["sed", "-i", "s/if has_moe_params:/if False: # CityQuest: dense nemotron_h/", | |
| f"{llama_dir}/conversion/nemotron.py"], | |
| check=True, | |
| ) | |
| f16 = f"{gguf_dir}/model-f16.gguf" | |
| subprocess.run( | |
| ["python", convert_py, merged_dir, "--outfile", f16, "--outtype", "f16"], | |
| check=True, | |
| ) | |
| out = f"{gguf_dir}/{GGUF_OUTFILE}" | |
| subprocess.run([quantize_bin, f16, out, "Q4_K_M"], check=True) | |
| return out | |
| def _model_card() -> str: | |
| return f"""--- | |
| base_model: {BASE_MODEL} | |
| library_name: gguf | |
| tags: | |
| - cityquest | |
| - nemotron | |
| - game-generation | |
| - llama.cpp | |
| --- | |
| # CityQuest Nemotron 3 Nano 4B (Q4_K_M GGUF) | |
| LoRA fine-tune of `{BASE_MODEL}` on the CityQuest location-based game dataset | |
| (~840 train / 92 val examples across scavenger hunt, hide-and-seek and tag). | |
| Trained to emit games directly in the app's `game_schema.json` contract | |
| (non-reasoning / JSON-only). Drop-in replacement for the stock Nemotron GGUF in | |
| `app/services/generator.py`. | |
| - LoRA r={LORA_R}, alpha={LORA_ALPHA}, epochs={NUM_EPOCHS}, lr={LEARNING_RATE} | |
| - Loaded via native `transformers` (arch `nemotron_h`); MoE router / lm_head excluded | |
| - Quantization: Q4_K_M · File: `{GGUF_OUTFILE}` | |
| """ | |
| def evaluate(repo: str, file: str, limit: int = 0) -> dict: | |
| """Run val prompts through a GGUF model and report JSON-parse / schema-pass rates. | |
| Runs on GPU in eval_image (llama.cpp), so it works even though the dev Codespace | |
| has no llama.cpp. Returns a metrics dict and logs a summary. | |
| """ | |
| import json | |
| import jsonschema | |
| from llama_cpp import Llama | |
| def extract_json(text: str): | |
| start = text.find("{") | |
| if start == -1: | |
| return None | |
| depth = 0 | |
| for i in range(start, len(text)): | |
| if text[i] == "{": | |
| depth += 1 | |
| elif text[i] == "}": | |
| depth -= 1 | |
| if depth == 0: | |
| raw = text[start:i + 1] | |
| if raw.startswith("{{") and raw.endswith("}}"): | |
| raw = raw[1:-1] | |
| return raw | |
| return None | |
| schema = json.loads(Path(REMOTE_SCHEMA).read_text()) | |
| rows = [json.loads(l) for l in Path(f"{REMOTE_DATA}/val.jsonl").read_text().splitlines() if l.strip()] | |
| if limit: | |
| rows = rows[:limit] | |
| print(f"[eval] {repo}/{file} — loading") | |
| llm = Llama.from_pretrained(repo_id=repo, filename=file, n_ctx=8192, n_gpu_layers=-1, verbose=False) | |
| parsed_ok = schema_ok = 0 | |
| for row in rows: | |
| messages = [m for m in row["messages"] if m["role"] in ("system", "user")] | |
| result = llm.create_chat_completion( | |
| messages=messages, max_tokens=8192, temperature=0.3, top_p=0.9, stop=["```"], | |
| ) | |
| text = result["choices"][0]["message"]["content"].strip() | |
| js = extract_json(text) | |
| if not js: | |
| continue | |
| try: | |
| game = json.loads(js) | |
| except json.JSONDecodeError: | |
| continue | |
| parsed_ok += 1 | |
| try: | |
| jsonschema.validate(instance=game, schema=schema) | |
| schema_ok += 1 | |
| except jsonschema.ValidationError: | |
| pass | |
| n = len(rows) | |
| metrics = { | |
| "model": f"{repo}/{file}", | |
| "n": n, | |
| "json_parse_rate": round(100 * parsed_ok / n, 1), | |
| "schema_pass_rate": round(100 * schema_ok / n, 1), | |
| } | |
| print(f"[eval] {metrics['model']}: parse={metrics['json_parse_rate']}% " | |
| f"schema={metrics['schema_pass_rate']}% (n={n})") | |
| return metrics | |
| def main(eval_baseline: bool = True, eval_limit: int = 0): | |
| """Train + upload, then evaluate the fine-tuned model (and baseline) on the val set.""" | |
| train.remote() | |
| print("\n=== Evaluation ===") | |
| tuned = evaluate.remote(HF_REPO, GGUF_OUTFILE, eval_limit) | |
| print(f"Fine-tuned : schema {tuned['schema_pass_rate']}% parse {tuned['json_parse_rate']}%") | |
| if eval_baseline: | |
| base = evaluate.remote(BASELINE_REPO, BASELINE_FILE, eval_limit) | |
| print(f"Baseline : schema {base['schema_pass_rate']}% parse {base['json_parse_rate']}%") | |
| verdict = ("PASS — safe to swap" if tuned["schema_pass_rate"] >= base["schema_pass_rate"] | |
| else "REGRESSION — do not swap") | |
| print(f"Gate: fine-tuned >= baseline ? {verdict}") | |
| def convert_only(): | |
| """Convert+upload from the committed merged model (no GPU, no peft). | |
| modal run --detach training/train_modal.py::convert_from_merged | |
| """ | |
| convert_from_merged.remote() | |
| def eval_only(repo: str = HF_REPO, file: str = GGUF_OUTFILE, limit: int = 0): | |
| """Evaluate an already-uploaded GGUF without retraining. | |
| Example: | |
| modal run training/train_modal.py::eval_only | |
| modal run training/train_modal.py::eval_only --repo nvidia/... --file ....gguf | |
| """ | |
| print(evaluate.remote(repo, file, limit)) | |