CityQuest-AI / training /train_modal.py
NANInithin
Claude Opus 4.8
feat(training): LoRA fine-tune pipeline for Nemotron 3 Nano 4B on Modal
7bb583b
Raw History Blame Contribute Delete
27.9 kB
"""Fine-tune NVIDIA Nemotron 3 Nano 4B on CityQuest data (LoRA) on Modal, then
merge, convert to GGUF Q4_K_M and upload to Hugging Face.
The model (`nemotron_h`, a hybrid Mamba-Transformer) is loaded **directly from
Hugging Face with plain `transformers` + PEFT** — no Unsloth (its model-type
detector doesn't recognise this brand-new architecture). transformers 5.x ships a
native NemotronH implementation with a pure-PyTorch fallback, so the CUDA Mamba
kernels (mamba-ssm) are optional; training just runs a bit slower without them.
Pipeline (one A100-80GB):
1. Load base (bf16) from HF, attach LoRA to all linear layers (router/lm_head excluded).
2. SFT on training/data/train.jsonl with response-only loss masking; eval on val.jsonl.
3. Merge LoRA → 16-bit safetensors (+ tokenizer & chat template).
4. Convert merged model → GGUF Q4_K_M via llama.cpp.
5. Upload the .gguf to HF_REPO so generator.py can pull it.
Run (after `modal token new` + the HF secret — see training/README.md):
modal run training/train_modal.py::main
"""
from __future__ import annotations
import os
from pathlib import Path
import modal
# ── Configuration ─────────────────────────────────────────────────────────────
BASE_MODEL = "nvidia/NVIDIA-Nemotron-3-Nano-4B-BF16" # official full-precision base on HF
HF_REPO = "NANI-Nithin/CityQuest-Nemotron-3-Nano-4B-GGUF"
GGUF_OUTFILE = "CityQuest-Nemotron-3-Nano-4B-Q4_K_M.gguf"
# llama.cpp commit whose convert_hf_to_gguf handles the DENSE nemotron_h 4B
# (earlier master misclassified it as MoE → KeyError 'moe_intermediate_size').
LLAMA_CPP_COMMIT = "e36a602ba38a26206c749ba4fb5dcf481bfd92db"
# Stock baseline GGUF (matches generator.py's current constants) for comparison.
BASELINE_REPO = "nvidia/NVIDIA-Nemotron-3-Nano-4B-GGUF"
BASELINE_FILE = "NVIDIA-Nemotron3-Nano-4B-Q4_K_M.gguf"
MAX_SEQ_LEN = 8192 # matches serve-time n_ctx; covers our longest sample (~5.4k)
NUM_EPOCHS = 3
LEARNING_RATE = 2e-4
LORA_R = 16
LORA_ALPHA = 16
BATCH_SIZE = 1
GRAD_ACCUM = 16 # effective batch 16
ASSISTANT_HEADER = "<|im_start|>assistant\n" # from the model's chat template
LOCAL_DATA = Path(__file__).resolve().parent / "data"
REMOTE_DATA = "/root/data"
LOCAL_SCHEMA = Path(__file__).resolve().parent.parent / "app" / "schemas" / "game_schema.json"
REMOTE_SCHEMA = "/root/game_schema.json"
# ── Training image ─────────────────────────────────────────────────────────
# CUDA-devel base (provides nvcc) so we can build the Mamba CUDA kernels. Without
# mamba-ssm/causal-conv1d the native nemotron_h falls back to a naive O(L^2) Mamba
# path that OOMs at our sequence lengths — the fused kernels are required here.
image = (
modal.Image.from_registry("nvidia/cuda:12.4.1-devel-ubuntu22.04", add_python="3.11")
.apt_install("git", "build-essential", "cmake", "curl", "libcurl4-openssl-dev")
.env({
"TORCH_CUDA_ARCH_LIST": "8.0", "MAX_JOBS": "4", "CC": "gcc", "CXX": "g++",
"CAUSAL_CONV1D_FORCE_BUILD": "TRUE", "MAMBA_FORCE_BUILD": "TRUE",
})
# Single controlled install pass. A constraints file pins torch==2.6.0 so NOTHING
# (transformers/peft/accelerate) can swap it for the PyPI cu130 build mid-install —
# that version drift is what broke the kernel ABI (undefined c10::cuda symbol).
# The Mamba kernels are force-compiled FROM SOURCE against this exact torch, then
# imported AT BUILD TIME so any ABI mismatch fails the build (CPU builder, ~1 min,
# no GPU burned) with the torch version printed.
.run_commands(
"pip install --upgrade pip",
"pip install torch==2.6.0 --index-url https://download.pytorch.org/whl/cu124",
"printf 'torch==2.6.0\\n' > /root/constraints.txt",
"pip install -c /root/constraints.txt ninja packaging wheel setuptools "
"'transformers>=5.4.0' 'peft>=0.13.0' 'accelerate>=1.0.0' 'datasets>=3.0.0' "
"huggingface_hub einops sentencepiece protobuf gguf 'numpy<2.0'",
"python -c \"import torch; print('[build] torch', torch.__version__, torch.version.cuda)\"",
"pip install -c /root/constraints.txt --no-build-isolation causal-conv1d",
# Pin 2.3.1: newer 2.3.2.post1 drags in tilelang/triton>=3.5 which conflict
# with torch 2.6's triton 3.2 (pip would backtrack to 2.3.1 anyway).
"pip install -c /root/constraints.txt --no-build-isolation mamba-ssm==2.3.1",
# transformers' lazy_load_kernel prefers the HF `kernels` hub package (no working
# mamba build → silent fallback to the OOM naive path). Remove it so transformers
# imports the mamba_ssm/causal_conv1d we just compiled.
"pip uninstall -y kernels || true",
# Build-time ABI gate: import the compiled CUDA C-extension only (this is what
# had the undefined-symbol error). It validates the torch ABI without triggering
# Triton autotune, which needs a GPU the CPU image-builder doesn't have.
"python -c \"import torch, selective_scan_cuda; "
"print('[build] CUDA ext ABI OK on torch', torch.__version__)\"",
)
.add_local_dir(str(LOCAL_DATA), remote_path=REMOTE_DATA)
)
# Eval image: CUDA *runtime* base so the cu124 llama-cpp-python wheel finds
# libcudart.so.12 (debian_slim lacks the CUDA runtime libs).
eval_image = (
modal.Image.from_registry("nvidia/cuda:12.4.1-runtime-ubuntu22.04", add_python="3.11")
.pip_install("huggingface_hub", "hf_transfer", "jsonschema")
.pip_install(
"llama-cpp-python",
extra_index_url="https://abetlen.github.io/llama-cpp-python/whl/cu124",
)
.env({"HF_HUB_ENABLE_HF_TRANSFER": "1"})
.add_local_dir(str(LOCAL_DATA), remote_path=REMOTE_DATA)
.add_local_file(str(LOCAL_SCHEMA), remote_path=REMOTE_SCHEMA)
)
app = modal.App("cityquest-nemotron-finetune", image=image)
hf_cache = modal.Volume.from_name("cityquest-hf-cache", create_if_missing=True)
outputs_vol = modal.Volume.from_name("cityquest-outputs", create_if_missing=True)
def _find_subsequence(seq: list[int], sub: list[int]) -> int:
"""Return the start index of the last occurrence of ``sub`` in ``seq``, else -1."""
if not sub:
return -1
for start in range(len(seq) - len(sub), -1, -1):
if seq[start:start + len(sub)] == sub:
return start
return -1
@app.function(
gpu="A100-80GB",
timeout=3 * 60 * 60,
secrets=[modal.Secret.from_name("huggingface")],
volumes={"/root/.cache/huggingface": hf_cache, "/root/outputs": outputs_vol},
)
def train():
import json
import os
import torch
import torch.nn as nn
from datasets import load_dataset
from peft import LoraConfig, get_peft_model
from transformers import (
AutoModelForCausalLM,
AutoTokenizer,
DataCollatorForSeq2Seq,
Trainer,
TrainerCallback,
TrainingArguments,
)
from transformers.trainer_utils import get_last_checkpoint
merged_dir = "/root/outputs/merged_16bit"
gguf_dir = "/root/outputs/gguf"
# ── 0. Mamba-kernel diagnostic + hard gate ─────────────────────────────
# The native nemotron_h falls back to a naive O(L^2) Mamba path (OOMs at our
# sequence lengths) unless mamba-ssm/causal-conv1d import cleanly. Surface the
# real reason and fail fast (seconds) instead of OOMing after model load.
print(f"[diag] torch={torch.__version__} cuda_avail={torch.cuda.is_available()} "
f"torch.cuda={torch.version.cuda}")
try:
import kernels # noqa: F401
print("[diag] WARNING: `kernels` package still installed (may hijack kernel loading)")
except ImportError:
print("[diag] `kernels` package absent (good — will import compiled mamba_ssm)")
from transformers.utils.import_utils import (
is_causal_conv1d_available, is_mamba_2_ssm_available,
)
print(f"[diag] is_mamba_2_ssm_available={is_mamba_2_ssm_available()} "
f"is_causal_conv1d_available={is_causal_conv1d_available()}")
_kernel_err = None
try:
import mamba_ssm
from mamba_ssm.ops.triton.ssd_combined import mamba_chunk_scan_combined # noqa: F401
from mamba_ssm.ops.triton.selective_state_update import selective_state_update # noqa: F401
import causal_conv1d # noqa: F401
from causal_conv1d import causal_conv1d_fn # noqa: F401
print(f"[diag] mamba_ssm {getattr(mamba_ssm, '__version__', '?')} + causal_conv1d import OK")
except Exception as e: # noqa: BLE001
import traceback
_kernel_err = traceback.format_exc()
print(f"[diag] KERNEL IMPORT FAILED:\n{_kernel_err}")
if _kernel_err is not None:
raise RuntimeError(
"Mamba CUDA kernels not importable — training would fall back to the "
"OOM-prone naive path. See [diag] traceback above for the root cause."
)
# ── 1. Load base + LoRA (native transformers, no Unsloth) ──────────────
# IMPORTANT: trust_remote_code=False → use transformers' *native* nemotron_h
# implementation, which falls back to a pure-PyTorch path when the mamba-ssm /
# causal-conv1d CUDA kernels are absent (it only warns). NVIDIA's *remote* code
# hard-raises ImportError without those kernels.
print(f"[train] loading base from HF: {BASE_MODEL}")
tokenizer = AutoTokenizer.from_pretrained(BASE_MODEL)
if tokenizer.pad_token is None:
tokenizer.pad_token = tokenizer.eos_token # this model ships no pad token
model = AutoModelForCausalLM.from_pretrained(
BASE_MODEL,
dtype=torch.bfloat16,
low_cpu_mem_usage=True,
attn_implementation="sdpa", # memory-efficient attention for the transformer layers
)
model.config.pad_token_id = tokenizer.pad_token_id
model.config.use_cache = False
model.gradient_checkpointing_enable(gradient_checkpointing_kwargs={"use_reentrant": False})
model.enable_input_require_grads()
# Discover LoRA targets: every nn.Linear leaf name except the LM head and any
# MoE router (router weights shouldn't be adapted). Covers attention + MLP +
# Mamba projections regardless of the exact arch naming.
target_modules = set()
for name, module in model.named_modules():
if isinstance(module, nn.Linear):
leaf = name.split(".")[-1]
if leaf in ("lm_head",) or "router" in name.lower():
continue
target_modules.add(leaf)
target_modules = sorted(target_modules)
print(f"[train] LoRA target modules: {target_modules}")
model = get_peft_model(model, LoraConfig(
r=LORA_R, lora_alpha=LORA_ALPHA, lora_dropout=0.0, bias="none",
task_type="CAUSAL_LM", target_modules=target_modules,
))
model.print_trainable_parameters()
# ── 2. Tokenize with response-only label masking ───────────────────────
header_ids = tokenizer(ASSISTANT_HEADER, add_special_tokens=False)["input_ids"]
def tokenize_and_mask(row):
# enable_thinking=False keeps non-reasoning/JSON-only mode (matches the
# serve-time "output only valid JSON" system prompt).
try:
text = tokenizer.apply_chat_template(
row["messages"], tokenize=False, add_generation_prompt=False,
enable_thinking=False,
)
except TypeError:
text = tokenizer.apply_chat_template(
row["messages"], tokenize=False, add_generation_prompt=False,
)
enc = tokenizer(text, truncation=True, max_length=MAX_SEQ_LEN, add_special_tokens=False)
input_ids = enc["input_ids"]
labels = list(input_ids)
cut = _find_subsequence(input_ids, header_ids)
if cut != -1:
for i in range(cut + len(header_ids)):
labels[i] = -100 # mask system + user; train only on the JSON answer
enc["labels"] = labels
return enc
train_ds = load_dataset("json", data_files=f"{REMOTE_DATA}/train.jsonl", split="train")
val_ds = load_dataset("json", data_files=f"{REMOTE_DATA}/val.jsonl", split="train")
train_ds = train_ds.map(tokenize_and_mask, remove_columns=train_ds.column_names)
val_ds = val_ds.map(tokenize_and_mask, remove_columns=val_ds.column_names)
print(f"[train] train={len(train_ds)} val={len(val_ds)}")
collator = DataCollatorForSeq2Seq(tokenizer, label_pad_token_id=-100, padding="longest")
# Checkpoint to the volume every few steps so a Modal preemption (spot GPU
# reclaim) resumes from the last step instead of restarting at 0. A callback
# commits the volume after each save so checkpoints survive an abrupt restart.
ckpt_dir = "/root/outputs/checkpoints"
class _CommitCheckpoint(TrainerCallback):
def on_save(self, args, state, control, **kwargs):
outputs_vol.commit()
print(f"[train] checkpoint committed @ step {state.global_step}")
trainer = Trainer(
model=model,
args=TrainingArguments(
output_dir=ckpt_dir,
per_device_train_batch_size=BATCH_SIZE,
gradient_accumulation_steps=GRAD_ACCUM,
warmup_ratio=0.05,
num_train_epochs=NUM_EPOCHS,
learning_rate=LEARNING_RATE,
lr_scheduler_type="cosine",
bf16=True,
logging_steps=10,
eval_strategy="epoch",
save_strategy="steps",
save_steps=20,
save_total_limit=1,
optim="adamw_torch",
seed=13,
report_to="none",
gradient_checkpointing=True,
gradient_checkpointing_kwargs={"use_reentrant": False},
),
train_dataset=train_ds,
eval_dataset=val_ds,
data_collator=collator,
callbacks=[_CommitCheckpoint()],
)
# ── 3. Train (resume from last checkpoint if a preemption left one) ─────
last_ckpt = get_last_checkpoint(ckpt_dir) if os.path.isdir(ckpt_dir) else None
if last_ckpt:
print(f"[train] resuming from checkpoint: {last_ckpt}")
stats = trainer.train(resume_from_checkpoint=last_ckpt)
print(f"[train] done: {stats.metrics}")
# ── 3b. Save the LoRA adapter first + commit (cheap recovery point) ─────
# Persist the trained adapter before the merge/GGUF steps so a later failure
# never costs us the ~40 min of training again.
adapter_dir = "/root/outputs/lora_adapter"
model.save_pretrained(adapter_dir)
tokenizer.save_pretrained(adapter_dir)
outputs_vol.commit()
print(f"[train] saved LoRA adapter → {adapter_dir} (committed)")
# ── 4. Merge LoRA → 16-bit, then convert to GGUF + upload ──────────────
print("[train] merging LoRA → 16-bit safetensors")
merged = model.merge_and_unload()
_save_convert_upload(merged, tokenizer, merged_dir, gguf_dir)
@app.function(
gpu="A100-80GB",
timeout=60 * 60,
secrets=[modal.Secret.from_name("huggingface")],
volumes={"/root/.cache/huggingface": hf_cache, "/root/outputs": outputs_vol},
)
def finalize():
"""Resume from the committed LoRA adapter: re-merge → GGUF → upload.
Lets us recover a completed training run without retraining (e.g. when only the
GGUF/upload tail failed). Needs a GPU because instantiating nemotron_h imports
the Triton Mamba kernels.
"""
import torch
from peft import PeftModel
from transformers import AutoModelForCausalLM, AutoTokenizer
adapter_dir = "/root/outputs/lora_adapter"
merged_dir = "/root/outputs/merged_16bit"
gguf_dir = "/root/outputs/gguf"
print(f"[finalize] loading base {BASE_MODEL} + adapter {adapter_dir}")
tokenizer = AutoTokenizer.from_pretrained(adapter_dir)
base = AutoModelForCausalLM.from_pretrained(
BASE_MODEL, dtype=torch.bfloat16, low_cpu_mem_usage=True, attn_implementation="sdpa",
)
model = PeftModel.from_pretrained(base, adapter_dir)
merged = model.merge_and_unload()
_save_convert_upload(merged, tokenizer, merged_dir, gguf_dir)
@app.function(
timeout=60 * 60,
secrets=[modal.Secret.from_name("huggingface")],
volumes={"/root/outputs": outputs_vol},
)
def convert_from_merged():
"""Convert an already-committed merged model → GGUF → upload. No GPU, no peft.
The fast/cheap resume path when train()'s GGUF/upload tail fails but the merged
model was committed first (it is, before the GGUF step). convert_hf_to_gguf reads
config + safetensors directly — no model instantiation, so no Mamba kernels/GPU.
"""
import os
from huggingface_hub import HfApi
merged_dir = "/root/outputs/merged_16bit"
gguf_dir = "/root/outputs/gguf"
_prepare_config_for_conversion(merged_dir)
gguf_path = _convert_with_llama_cpp(merged_dir, gguf_dir)
outputs_vol.commit()
api = HfApi(token=os.environ["HF_TOKEN"])
api.create_repo(HF_REPO, repo_type="model", exist_ok=True)
api.upload_file(path_or_fileobj=gguf_path, path_in_repo=GGUF_OUTFILE,
repo_id=HF_REPO, repo_type="model")
api.upload_file(path_or_fileobj=_model_card().encode("utf-8"), path_in_repo="README.md",
repo_id=HF_REPO, repo_type="model")
print(f"[convert] uploaded {GGUF_OUTFILE} → https://huggingface.co/{HF_REPO}")
def _prepare_config_for_conversion(model_dir: str) -> None:
"""Overwrite the merged config.json with the *original base* config — verbatim,
including ``auto_map`` — so llama.cpp's converter reads the right hparams.
The converter's ``load_hparams`` does ``AutoConfig.from_pretrained(
trust_remote_code=False).to_dict()``. For nemotron_h that returns NemotronHConfig
with its *class-default* MoE attributes (``num_experts_per_tok``), which makes the
converter think the dense 4B is MoE → ``KeyError: moe_intermediate_size``.
Keeping ``auto_map`` makes that AutoConfig call refuse the custom code and fall
back to the *raw* config.json — the base config, which has ``num_hidden_layers``
and ``intermediate_size`` but NO expert keys → converter takes the dense path.
(transformers 5.12's own re-save dropped ``num_hidden_layers``, so we must use the
base config, not the merged one.)
"""
import json
from huggingface_hub import hf_hub_download
base_cfg = hf_hub_download(BASE_MODEL, "config.json")
data = json.loads(Path(base_cfg).read_text()) # keep auto_map intact on purpose
(Path(model_dir) / "config.json").write_text(json.dumps(data, indent=2))
print(f"[finalize] wrote verbatim base config.json (auto_map kept → raw-JSON fallback) → {model_dir}")
def _save_convert_upload(merged, tokenizer, merged_dir: str, gguf_dir: str) -> str:
"""Normalize config, save merged model, convert to GGUF Q4_K_M, upload to HF.
Shared by ``train`` (in-memory merged model) and ``finalize`` (re-merged from
the committed LoRA adapter).
"""
import os
from huggingface_hub import HfApi
# Nemotron ships generation_config with top_p set but do_sample=False, which
# transformers' strict save-validation rejects. Normalize to clean greedy.
gc = merged.generation_config
gc.do_sample = False
for _attr in ("top_p", "top_k", "typical_p", "temperature"):
if getattr(gc, _attr, None) is not None:
setattr(gc, _attr, None)
merged.save_pretrained(merged_dir, safe_serialization=True)
tokenizer.save_pretrained(merged_dir)
_prepare_config_for_conversion(merged_dir)
outputs_vol.commit() # persist merged model before the (fragile) GGUF step
gguf_path = _convert_with_llama_cpp(merged_dir, gguf_dir)
print(f"[finalize] GGUF ready: {gguf_path}")
outputs_vol.commit()
api = HfApi(token=os.environ["HF_TOKEN"])
api.create_repo(HF_REPO, repo_type="model", exist_ok=True)
api.upload_file(path_or_fileobj=gguf_path, path_in_repo=GGUF_OUTFILE,
repo_id=HF_REPO, repo_type="model")
api.upload_file(path_or_fileobj=_model_card().encode("utf-8"), path_in_repo="README.md",
repo_id=HF_REPO, repo_type="model")
print(f"[finalize] uploaded {GGUF_OUTFILE} → https://huggingface.co/{HF_REPO}")
return gguf_path
def _convert_with_llama_cpp(merged_dir: str, gguf_dir: str) -> str:
"""Clone llama.cpp, convert merged HF model to f16 GGUF, quantize Q4_K_M.
Uses latest llama.cpp, which supports the nemotron_h architecture (NVIDIA and
others publish Nemotron-3 GGUFs built with it).
"""
import shutil
import subprocess
Path(gguf_dir).mkdir(parents=True, exist_ok=True)
# Commit-specific path so a stale clone in a reused (warm) Modal container can't
# be mistaken for the pinned version — that bug skipped the checkout and used an
# old converter that misclassified the dense 4B as MoE.
llama_dir = f"/root/llama.cpp-{LLAMA_CPP_COMMIT[:8]}"
quantize_bin = f"{llama_dir}/build/bin/llama-quantize"
convert_py = f"{llama_dir}/convert_hf_to_gguf.py"
if not Path(quantize_bin).exists():
if Path(llama_dir).exists():
shutil.rmtree(llama_dir)
subprocess.run(["git", "clone", "https://github.com/ggml-org/llama.cpp", llama_dir], check=True)
# Pin to a commit whose converter has the nemotron_h dense/MoE split.
subprocess.run(["git", "-C", llama_dir, "checkout", LLAMA_CPP_COMMIT], check=True)
# NOTE: deliberately NOT `pip install -r requirements.txt` — it downgrades our
# transformers below nemotron_h support. The convert deps are already in the image.
subprocess.run(["cmake", "-S", llama_dir, "-B", f"{llama_dir}/build", "-DLLAMA_CURL=OFF"], check=True)
subprocess.run(["cmake", "--build", f"{llama_dir}/build", "--target", "llama-quantize", "-j"], check=True)
# Force the DENSE path (idempotent — run every call, NOT just on fresh clone, since
# a warm container may already have the built dir and skip the block above).
# transformers' NemotronHConfig always carries a default num_experts_per_tok=2, so
# the converter's `if has_moe_params` always trips and treats our dense 4B as MoE
# (→ KeyError moe_intermediate_size, and a broken empty-FFN GGUF). The 4B has no
# experts, so disabling that branch is correct.
subprocess.run(
["sed", "-i", "s/if has_moe_params:/if False: # CityQuest: dense nemotron_h/",
f"{llama_dir}/conversion/nemotron.py"],
check=True,
)
f16 = f"{gguf_dir}/model-f16.gguf"
subprocess.run(
["python", convert_py, merged_dir, "--outfile", f16, "--outtype", "f16"],
check=True,
)
out = f"{gguf_dir}/{GGUF_OUTFILE}"
subprocess.run([quantize_bin, f16, out, "Q4_K_M"], check=True)
return out
def _model_card() -> str:
return f"""---
base_model: {BASE_MODEL}
library_name: gguf
tags:
- cityquest
- nemotron
- game-generation
- llama.cpp
---
# CityQuest Nemotron 3 Nano 4B (Q4_K_M GGUF)
LoRA fine-tune of `{BASE_MODEL}` on the CityQuest location-based game dataset
(~840 train / 92 val examples across scavenger hunt, hide-and-seek and tag).
Trained to emit games directly in the app's `game_schema.json` contract
(non-reasoning / JSON-only). Drop-in replacement for the stock Nemotron GGUF in
`app/services/generator.py`.
- LoRA r={LORA_R}, alpha={LORA_ALPHA}, epochs={NUM_EPOCHS}, lr={LEARNING_RATE}
- Loaded via native `transformers` (arch `nemotron_h`); MoE router / lm_head excluded
- Quantization: Q4_K_M · File: `{GGUF_OUTFILE}`
"""
@app.function(
image=eval_image,
gpu="A100",
timeout=60 * 60,
secrets=[modal.Secret.from_name("huggingface")],
volumes={"/root/.cache/huggingface": hf_cache},
)
def evaluate(repo: str, file: str, limit: int = 0) -> dict:
"""Run val prompts through a GGUF model and report JSON-parse / schema-pass rates.
Runs on GPU in eval_image (llama.cpp), so it works even though the dev Codespace
has no llama.cpp. Returns a metrics dict and logs a summary.
"""
import json
import jsonschema
from llama_cpp import Llama
def extract_json(text: str):
start = text.find("{")
if start == -1:
return None
depth = 0
for i in range(start, len(text)):
if text[i] == "{":
depth += 1
elif text[i] == "}":
depth -= 1
if depth == 0:
raw = text[start:i + 1]
if raw.startswith("{{") and raw.endswith("}}"):
raw = raw[1:-1]
return raw
return None
schema = json.loads(Path(REMOTE_SCHEMA).read_text())
rows = [json.loads(l) for l in Path(f"{REMOTE_DATA}/val.jsonl").read_text().splitlines() if l.strip()]
if limit:
rows = rows[:limit]
print(f"[eval] {repo}/{file} — loading")
llm = Llama.from_pretrained(repo_id=repo, filename=file, n_ctx=8192, n_gpu_layers=-1, verbose=False)
parsed_ok = schema_ok = 0
for row in rows:
messages = [m for m in row["messages"] if m["role"] in ("system", "user")]
result = llm.create_chat_completion(
messages=messages, max_tokens=8192, temperature=0.3, top_p=0.9, stop=["```"],
)
text = result["choices"][0]["message"]["content"].strip()
js = extract_json(text)
if not js:
continue
try:
game = json.loads(js)
except json.JSONDecodeError:
continue
parsed_ok += 1
try:
jsonschema.validate(instance=game, schema=schema)
schema_ok += 1
except jsonschema.ValidationError:
pass
n = len(rows)
metrics = {
"model": f"{repo}/{file}",
"n": n,
"json_parse_rate": round(100 * parsed_ok / n, 1),
"schema_pass_rate": round(100 * schema_ok / n, 1),
}
print(f"[eval] {metrics['model']}: parse={metrics['json_parse_rate']}% "
f"schema={metrics['schema_pass_rate']}% (n={n})")
return metrics
@app.local_entrypoint()
def main(eval_baseline: bool = True, eval_limit: int = 0):
"""Train + upload, then evaluate the fine-tuned model (and baseline) on the val set."""
train.remote()
print("\n=== Evaluation ===")
tuned = evaluate.remote(HF_REPO, GGUF_OUTFILE, eval_limit)
print(f"Fine-tuned : schema {tuned['schema_pass_rate']}% parse {tuned['json_parse_rate']}%")
if eval_baseline:
base = evaluate.remote(BASELINE_REPO, BASELINE_FILE, eval_limit)
print(f"Baseline : schema {base['schema_pass_rate']}% parse {base['json_parse_rate']}%")
verdict = ("PASS — safe to swap" if tuned["schema_pass_rate"] >= base["schema_pass_rate"]
else "REGRESSION — do not swap")
print(f"Gate: fine-tuned >= baseline ? {verdict}")
@app.local_entrypoint()
def convert_only():
"""Convert+upload from the committed merged model (no GPU, no peft).
modal run --detach training/train_modal.py::convert_from_merged
"""
convert_from_merged.remote()
@app.local_entrypoint()
def eval_only(repo: str = HF_REPO, file: str = GGUF_OUTFILE, limit: int = 0):
"""Evaluate an already-uploaded GGUF without retraining.
Example:
modal run training/train_modal.py::eval_only
modal run training/train_modal.py::eval_only --repo nvidia/... --file ....gguf
"""
print(evaluate.remote(repo, file, limit))