"""Fine-tune NVIDIA Nemotron 3 Nano 4B on CityQuest data (LoRA) on Modal, then merge, convert to GGUF Q4_K_M and upload to Hugging Face. The model (`nemotron_h`, a hybrid Mamba-Transformer) is loaded **directly from Hugging Face with plain `transformers` + PEFT** — no Unsloth (its model-type detector doesn't recognise this brand-new architecture). transformers 5.x ships a native NemotronH implementation with a pure-PyTorch fallback, so the CUDA Mamba kernels (mamba-ssm) are optional; training just runs a bit slower without them. Pipeline (one A100-80GB): 1. Load base (bf16) from HF, attach LoRA to all linear layers (router/lm_head excluded). 2. SFT on training/data/train.jsonl with response-only loss masking; eval on val.jsonl. 3. Merge LoRA → 16-bit safetensors (+ tokenizer & chat template). 4. Convert merged model → GGUF Q4_K_M via llama.cpp. 5. Upload the .gguf to HF_REPO so generator.py can pull it. Run (after `modal token new` + the HF secret — see training/README.md): modal run training/train_modal.py::main """ from __future__ import annotations import os from pathlib import Path import modal # ── Configuration ───────────────────────────────────────────────────────────── BASE_MODEL = "nvidia/NVIDIA-Nemotron-3-Nano-4B-BF16" # official full-precision base on HF HF_REPO = "NANI-Nithin/CityQuest-Nemotron-3-Nano-4B-GGUF" GGUF_OUTFILE = "CityQuest-Nemotron-3-Nano-4B-Q4_K_M.gguf" # llama.cpp commit whose convert_hf_to_gguf handles the DENSE nemotron_h 4B # (earlier master misclassified it as MoE → KeyError 'moe_intermediate_size'). LLAMA_CPP_COMMIT = "e36a602ba38a26206c749ba4fb5dcf481bfd92db" # Stock baseline GGUF (matches generator.py's current constants) for comparison. BASELINE_REPO = "nvidia/NVIDIA-Nemotron-3-Nano-4B-GGUF" BASELINE_FILE = "NVIDIA-Nemotron3-Nano-4B-Q4_K_M.gguf" MAX_SEQ_LEN = 8192 # matches serve-time n_ctx; covers our longest sample (~5.4k) NUM_EPOCHS = 3 LEARNING_RATE = 2e-4 LORA_R = 16 LORA_ALPHA = 16 BATCH_SIZE = 1 GRAD_ACCUM = 16 # effective batch 16 ASSISTANT_HEADER = "<|im_start|>assistant\n" # from the model's chat template LOCAL_DATA = Path(__file__).resolve().parent / "data" REMOTE_DATA = "/root/data" LOCAL_SCHEMA = Path(__file__).resolve().parent.parent / "app" / "schemas" / "game_schema.json" REMOTE_SCHEMA = "/root/game_schema.json" # ── Training image ───────────────────────────────────────────────────────── # CUDA-devel base (provides nvcc) so we can build the Mamba CUDA kernels. Without # mamba-ssm/causal-conv1d the native nemotron_h falls back to a naive O(L^2) Mamba # path that OOMs at our sequence lengths — the fused kernels are required here. image = ( modal.Image.from_registry("nvidia/cuda:12.4.1-devel-ubuntu22.04", add_python="3.11") .apt_install("git", "build-essential", "cmake", "curl", "libcurl4-openssl-dev") .env({ "TORCH_CUDA_ARCH_LIST": "8.0", "MAX_JOBS": "4", "CC": "gcc", "CXX": "g++", "CAUSAL_CONV1D_FORCE_BUILD": "TRUE", "MAMBA_FORCE_BUILD": "TRUE", }) # Single controlled install pass. A constraints file pins torch==2.6.0 so NOTHING # (transformers/peft/accelerate) can swap it for the PyPI cu130 build mid-install — # that version drift is what broke the kernel ABI (undefined c10::cuda symbol). # The Mamba kernels are force-compiled FROM SOURCE against this exact torch, then # imported AT BUILD TIME so any ABI mismatch fails the build (CPU builder, ~1 min, # no GPU burned) with the torch version printed. .run_commands( "pip install --upgrade pip", "pip install torch==2.6.0 --index-url https://download.pytorch.org/whl/cu124", "printf 'torch==2.6.0\\n' > /root/constraints.txt", "pip install -c /root/constraints.txt ninja packaging wheel setuptools " "'transformers>=5.4.0' 'peft>=0.13.0' 'accelerate>=1.0.0' 'datasets>=3.0.0' " "huggingface_hub einops sentencepiece protobuf gguf 'numpy<2.0'", "python -c \"import torch; print('[build] torch', torch.__version__, torch.version.cuda)\"", "pip install -c /root/constraints.txt --no-build-isolation causal-conv1d", # Pin 2.3.1: newer 2.3.2.post1 drags in tilelang/triton>=3.5 which conflict # with torch 2.6's triton 3.2 (pip would backtrack to 2.3.1 anyway). "pip install -c /root/constraints.txt --no-build-isolation mamba-ssm==2.3.1", # transformers' lazy_load_kernel prefers the HF `kernels` hub package (no working # mamba build → silent fallback to the OOM naive path). Remove it so transformers # imports the mamba_ssm/causal_conv1d we just compiled. "pip uninstall -y kernels || true", # Build-time ABI gate: import the compiled CUDA C-extension only (this is what # had the undefined-symbol error). It validates the torch ABI without triggering # Triton autotune, which needs a GPU the CPU image-builder doesn't have. "python -c \"import torch, selective_scan_cuda; " "print('[build] CUDA ext ABI OK on torch', torch.__version__)\"", ) .add_local_dir(str(LOCAL_DATA), remote_path=REMOTE_DATA) ) # Eval image: CUDA *runtime* base so the cu124 llama-cpp-python wheel finds # libcudart.so.12 (debian_slim lacks the CUDA runtime libs). eval_image = ( modal.Image.from_registry("nvidia/cuda:12.4.1-runtime-ubuntu22.04", add_python="3.11") .pip_install("huggingface_hub", "hf_transfer", "jsonschema") .pip_install( "llama-cpp-python", extra_index_url="https://abetlen.github.io/llama-cpp-python/whl/cu124", ) .env({"HF_HUB_ENABLE_HF_TRANSFER": "1"}) .add_local_dir(str(LOCAL_DATA), remote_path=REMOTE_DATA) .add_local_file(str(LOCAL_SCHEMA), remote_path=REMOTE_SCHEMA) ) app = modal.App("cityquest-nemotron-finetune", image=image) hf_cache = modal.Volume.from_name("cityquest-hf-cache", create_if_missing=True) outputs_vol = modal.Volume.from_name("cityquest-outputs", create_if_missing=True) def _find_subsequence(seq: list[int], sub: list[int]) -> int: """Return the start index of the last occurrence of ``sub`` in ``seq``, else -1.""" if not sub: return -1 for start in range(len(seq) - len(sub), -1, -1): if seq[start:start + len(sub)] == sub: return start return -1 @app.function( gpu="A100-80GB", timeout=3 * 60 * 60, secrets=[modal.Secret.from_name("huggingface")], volumes={"/root/.cache/huggingface": hf_cache, "/root/outputs": outputs_vol}, ) def train(): import json import os import torch import torch.nn as nn from datasets import load_dataset from peft import LoraConfig, get_peft_model from transformers import ( AutoModelForCausalLM, AutoTokenizer, DataCollatorForSeq2Seq, Trainer, TrainerCallback, TrainingArguments, ) from transformers.trainer_utils import get_last_checkpoint merged_dir = "/root/outputs/merged_16bit" gguf_dir = "/root/outputs/gguf" # ── 0. Mamba-kernel diagnostic + hard gate ───────────────────────────── # The native nemotron_h falls back to a naive O(L^2) Mamba path (OOMs at our # sequence lengths) unless mamba-ssm/causal-conv1d import cleanly. Surface the # real reason and fail fast (seconds) instead of OOMing after model load. print(f"[diag] torch={torch.__version__} cuda_avail={torch.cuda.is_available()} " f"torch.cuda={torch.version.cuda}") try: import kernels # noqa: F401 print("[diag] WARNING: `kernels` package still installed (may hijack kernel loading)") except ImportError: print("[diag] `kernels` package absent (good — will import compiled mamba_ssm)") from transformers.utils.import_utils import ( is_causal_conv1d_available, is_mamba_2_ssm_available, ) print(f"[diag] is_mamba_2_ssm_available={is_mamba_2_ssm_available()} " f"is_causal_conv1d_available={is_causal_conv1d_available()}") _kernel_err = None try: import mamba_ssm from mamba_ssm.ops.triton.ssd_combined import mamba_chunk_scan_combined # noqa: F401 from mamba_ssm.ops.triton.selective_state_update import selective_state_update # noqa: F401 import causal_conv1d # noqa: F401 from causal_conv1d import causal_conv1d_fn # noqa: F401 print(f"[diag] mamba_ssm {getattr(mamba_ssm, '__version__', '?')} + causal_conv1d import OK") except Exception as e: # noqa: BLE001 import traceback _kernel_err = traceback.format_exc() print(f"[diag] KERNEL IMPORT FAILED:\n{_kernel_err}") if _kernel_err is not None: raise RuntimeError( "Mamba CUDA kernels not importable — training would fall back to the " "OOM-prone naive path. See [diag] traceback above for the root cause." ) # ── 1. Load base + LoRA (native transformers, no Unsloth) ────────────── # IMPORTANT: trust_remote_code=False → use transformers' *native* nemotron_h # implementation, which falls back to a pure-PyTorch path when the mamba-ssm / # causal-conv1d CUDA kernels are absent (it only warns). NVIDIA's *remote* code # hard-raises ImportError without those kernels. print(f"[train] loading base from HF: {BASE_MODEL}") tokenizer = AutoTokenizer.from_pretrained(BASE_MODEL) if tokenizer.pad_token is None: tokenizer.pad_token = tokenizer.eos_token # this model ships no pad token model = AutoModelForCausalLM.from_pretrained( BASE_MODEL, dtype=torch.bfloat16, low_cpu_mem_usage=True, attn_implementation="sdpa", # memory-efficient attention for the transformer layers ) model.config.pad_token_id = tokenizer.pad_token_id model.config.use_cache = False model.gradient_checkpointing_enable(gradient_checkpointing_kwargs={"use_reentrant": False}) model.enable_input_require_grads() # Discover LoRA targets: every nn.Linear leaf name except the LM head and any # MoE router (router weights shouldn't be adapted). Covers attention + MLP + # Mamba projections regardless of the exact arch naming. target_modules = set() for name, module in model.named_modules(): if isinstance(module, nn.Linear): leaf = name.split(".")[-1] if leaf in ("lm_head",) or "router" in name.lower(): continue target_modules.add(leaf) target_modules = sorted(target_modules) print(f"[train] LoRA target modules: {target_modules}") model = get_peft_model(model, LoraConfig( r=LORA_R, lora_alpha=LORA_ALPHA, lora_dropout=0.0, bias="none", task_type="CAUSAL_LM", target_modules=target_modules, )) model.print_trainable_parameters() # ── 2. Tokenize with response-only label masking ─────────────────────── header_ids = tokenizer(ASSISTANT_HEADER, add_special_tokens=False)["input_ids"] def tokenize_and_mask(row): # enable_thinking=False keeps non-reasoning/JSON-only mode (matches the # serve-time "output only valid JSON" system prompt). try: text = tokenizer.apply_chat_template( row["messages"], tokenize=False, add_generation_prompt=False, enable_thinking=False, ) except TypeError: text = tokenizer.apply_chat_template( row["messages"], tokenize=False, add_generation_prompt=False, ) enc = tokenizer(text, truncation=True, max_length=MAX_SEQ_LEN, add_special_tokens=False) input_ids = enc["input_ids"] labels = list(input_ids) cut = _find_subsequence(input_ids, header_ids) if cut != -1: for i in range(cut + len(header_ids)): labels[i] = -100 # mask system + user; train only on the JSON answer enc["labels"] = labels return enc train_ds = load_dataset("json", data_files=f"{REMOTE_DATA}/train.jsonl", split="train") val_ds = load_dataset("json", data_files=f"{REMOTE_DATA}/val.jsonl", split="train") train_ds = train_ds.map(tokenize_and_mask, remove_columns=train_ds.column_names) val_ds = val_ds.map(tokenize_and_mask, remove_columns=val_ds.column_names) print(f"[train] train={len(train_ds)} val={len(val_ds)}") collator = DataCollatorForSeq2Seq(tokenizer, label_pad_token_id=-100, padding="longest") # Checkpoint to the volume every few steps so a Modal preemption (spot GPU # reclaim) resumes from the last step instead of restarting at 0. A callback # commits the volume after each save so checkpoints survive an abrupt restart. ckpt_dir = "/root/outputs/checkpoints" class _CommitCheckpoint(TrainerCallback): def on_save(self, args, state, control, **kwargs): outputs_vol.commit() print(f"[train] checkpoint committed @ step {state.global_step}") trainer = Trainer( model=model, args=TrainingArguments( output_dir=ckpt_dir, per_device_train_batch_size=BATCH_SIZE, gradient_accumulation_steps=GRAD_ACCUM, warmup_ratio=0.05, num_train_epochs=NUM_EPOCHS, learning_rate=LEARNING_RATE, lr_scheduler_type="cosine", bf16=True, logging_steps=10, eval_strategy="epoch", save_strategy="steps", save_steps=20, save_total_limit=1, optim="adamw_torch", seed=13, report_to="none", gradient_checkpointing=True, gradient_checkpointing_kwargs={"use_reentrant": False}, ), train_dataset=train_ds, eval_dataset=val_ds, data_collator=collator, callbacks=[_CommitCheckpoint()], ) # ── 3. Train (resume from last checkpoint if a preemption left one) ───── last_ckpt = get_last_checkpoint(ckpt_dir) if os.path.isdir(ckpt_dir) else None if last_ckpt: print(f"[train] resuming from checkpoint: {last_ckpt}") stats = trainer.train(resume_from_checkpoint=last_ckpt) print(f"[train] done: {stats.metrics}") # ── 3b. Save the LoRA adapter first + commit (cheap recovery point) ───── # Persist the trained adapter before the merge/GGUF steps so a later failure # never costs us the ~40 min of training again. adapter_dir = "/root/outputs/lora_adapter" model.save_pretrained(adapter_dir) tokenizer.save_pretrained(adapter_dir) outputs_vol.commit() print(f"[train] saved LoRA adapter → {adapter_dir} (committed)") # ── 4. Merge LoRA → 16-bit, then convert to GGUF + upload ────────────── print("[train] merging LoRA → 16-bit safetensors") merged = model.merge_and_unload() _save_convert_upload(merged, tokenizer, merged_dir, gguf_dir) @app.function( gpu="A100-80GB", timeout=60 * 60, secrets=[modal.Secret.from_name("huggingface")], volumes={"/root/.cache/huggingface": hf_cache, "/root/outputs": outputs_vol}, ) def finalize(): """Resume from the committed LoRA adapter: re-merge → GGUF → upload. Lets us recover a completed training run without retraining (e.g. when only the GGUF/upload tail failed). Needs a GPU because instantiating nemotron_h imports the Triton Mamba kernels. """ import torch from peft import PeftModel from transformers import AutoModelForCausalLM, AutoTokenizer adapter_dir = "/root/outputs/lora_adapter" merged_dir = "/root/outputs/merged_16bit" gguf_dir = "/root/outputs/gguf" print(f"[finalize] loading base {BASE_MODEL} + adapter {adapter_dir}") tokenizer = AutoTokenizer.from_pretrained(adapter_dir) base = AutoModelForCausalLM.from_pretrained( BASE_MODEL, dtype=torch.bfloat16, low_cpu_mem_usage=True, attn_implementation="sdpa", ) model = PeftModel.from_pretrained(base, adapter_dir) merged = model.merge_and_unload() _save_convert_upload(merged, tokenizer, merged_dir, gguf_dir) @app.function( timeout=60 * 60, secrets=[modal.Secret.from_name("huggingface")], volumes={"/root/outputs": outputs_vol}, ) def convert_from_merged(): """Convert an already-committed merged model → GGUF → upload. No GPU, no peft. The fast/cheap resume path when train()'s GGUF/upload tail fails but the merged model was committed first (it is, before the GGUF step). convert_hf_to_gguf reads config + safetensors directly — no model instantiation, so no Mamba kernels/GPU. """ import os from huggingface_hub import HfApi merged_dir = "/root/outputs/merged_16bit" gguf_dir = "/root/outputs/gguf" _prepare_config_for_conversion(merged_dir) gguf_path = _convert_with_llama_cpp(merged_dir, gguf_dir) outputs_vol.commit() api = HfApi(token=os.environ["HF_TOKEN"]) api.create_repo(HF_REPO, repo_type="model", exist_ok=True) api.upload_file(path_or_fileobj=gguf_path, path_in_repo=GGUF_OUTFILE, repo_id=HF_REPO, repo_type="model") api.upload_file(path_or_fileobj=_model_card().encode("utf-8"), path_in_repo="README.md", repo_id=HF_REPO, repo_type="model") print(f"[convert] uploaded {GGUF_OUTFILE} → https://huggingface.co/{HF_REPO}") def _prepare_config_for_conversion(model_dir: str) -> None: """Overwrite the merged config.json with the *original base* config — verbatim, including ``auto_map`` — so llama.cpp's converter reads the right hparams. The converter's ``load_hparams`` does ``AutoConfig.from_pretrained( trust_remote_code=False).to_dict()``. For nemotron_h that returns NemotronHConfig with its *class-default* MoE attributes (``num_experts_per_tok``), which makes the converter think the dense 4B is MoE → ``KeyError: moe_intermediate_size``. Keeping ``auto_map`` makes that AutoConfig call refuse the custom code and fall back to the *raw* config.json — the base config, which has ``num_hidden_layers`` and ``intermediate_size`` but NO expert keys → converter takes the dense path. (transformers 5.12's own re-save dropped ``num_hidden_layers``, so we must use the base config, not the merged one.) """ import json from huggingface_hub import hf_hub_download base_cfg = hf_hub_download(BASE_MODEL, "config.json") data = json.loads(Path(base_cfg).read_text()) # keep auto_map intact on purpose (Path(model_dir) / "config.json").write_text(json.dumps(data, indent=2)) print(f"[finalize] wrote verbatim base config.json (auto_map kept → raw-JSON fallback) → {model_dir}") def _save_convert_upload(merged, tokenizer, merged_dir: str, gguf_dir: str) -> str: """Normalize config, save merged model, convert to GGUF Q4_K_M, upload to HF. Shared by ``train`` (in-memory merged model) and ``finalize`` (re-merged from the committed LoRA adapter). """ import os from huggingface_hub import HfApi # Nemotron ships generation_config with top_p set but do_sample=False, which # transformers' strict save-validation rejects. Normalize to clean greedy. gc = merged.generation_config gc.do_sample = False for _attr in ("top_p", "top_k", "typical_p", "temperature"): if getattr(gc, _attr, None) is not None: setattr(gc, _attr, None) merged.save_pretrained(merged_dir, safe_serialization=True) tokenizer.save_pretrained(merged_dir) _prepare_config_for_conversion(merged_dir) outputs_vol.commit() # persist merged model before the (fragile) GGUF step gguf_path = _convert_with_llama_cpp(merged_dir, gguf_dir) print(f"[finalize] GGUF ready: {gguf_path}") outputs_vol.commit() api = HfApi(token=os.environ["HF_TOKEN"]) api.create_repo(HF_REPO, repo_type="model", exist_ok=True) api.upload_file(path_or_fileobj=gguf_path, path_in_repo=GGUF_OUTFILE, repo_id=HF_REPO, repo_type="model") api.upload_file(path_or_fileobj=_model_card().encode("utf-8"), path_in_repo="README.md", repo_id=HF_REPO, repo_type="model") print(f"[finalize] uploaded {GGUF_OUTFILE} → https://huggingface.co/{HF_REPO}") return gguf_path def _convert_with_llama_cpp(merged_dir: str, gguf_dir: str) -> str: """Clone llama.cpp, convert merged HF model to f16 GGUF, quantize Q4_K_M. Uses latest llama.cpp, which supports the nemotron_h architecture (NVIDIA and others publish Nemotron-3 GGUFs built with it). """ import shutil import subprocess Path(gguf_dir).mkdir(parents=True, exist_ok=True) # Commit-specific path so a stale clone in a reused (warm) Modal container can't # be mistaken for the pinned version — that bug skipped the checkout and used an # old converter that misclassified the dense 4B as MoE. llama_dir = f"/root/llama.cpp-{LLAMA_CPP_COMMIT[:8]}" quantize_bin = f"{llama_dir}/build/bin/llama-quantize" convert_py = f"{llama_dir}/convert_hf_to_gguf.py" if not Path(quantize_bin).exists(): if Path(llama_dir).exists(): shutil.rmtree(llama_dir) subprocess.run(["git", "clone", "https://github.com/ggml-org/llama.cpp", llama_dir], check=True) # Pin to a commit whose converter has the nemotron_h dense/MoE split. subprocess.run(["git", "-C", llama_dir, "checkout", LLAMA_CPP_COMMIT], check=True) # NOTE: deliberately NOT `pip install -r requirements.txt` — it downgrades our # transformers below nemotron_h support. The convert deps are already in the image. subprocess.run(["cmake", "-S", llama_dir, "-B", f"{llama_dir}/build", "-DLLAMA_CURL=OFF"], check=True) subprocess.run(["cmake", "--build", f"{llama_dir}/build", "--target", "llama-quantize", "-j"], check=True) # Force the DENSE path (idempotent — run every call, NOT just on fresh clone, since # a warm container may already have the built dir and skip the block above). # transformers' NemotronHConfig always carries a default num_experts_per_tok=2, so # the converter's `if has_moe_params` always trips and treats our dense 4B as MoE # (→ KeyError moe_intermediate_size, and a broken empty-FFN GGUF). The 4B has no # experts, so disabling that branch is correct. subprocess.run( ["sed", "-i", "s/if has_moe_params:/if False: # CityQuest: dense nemotron_h/", f"{llama_dir}/conversion/nemotron.py"], check=True, ) f16 = f"{gguf_dir}/model-f16.gguf" subprocess.run( ["python", convert_py, merged_dir, "--outfile", f16, "--outtype", "f16"], check=True, ) out = f"{gguf_dir}/{GGUF_OUTFILE}" subprocess.run([quantize_bin, f16, out, "Q4_K_M"], check=True) return out def _model_card() -> str: return f"""--- base_model: {BASE_MODEL} library_name: gguf tags: - cityquest - nemotron - game-generation - llama.cpp --- # CityQuest Nemotron 3 Nano 4B (Q4_K_M GGUF) LoRA fine-tune of `{BASE_MODEL}` on the CityQuest location-based game dataset (~840 train / 92 val examples across scavenger hunt, hide-and-seek and tag). Trained to emit games directly in the app's `game_schema.json` contract (non-reasoning / JSON-only). Drop-in replacement for the stock Nemotron GGUF in `app/services/generator.py`. - LoRA r={LORA_R}, alpha={LORA_ALPHA}, epochs={NUM_EPOCHS}, lr={LEARNING_RATE} - Loaded via native `transformers` (arch `nemotron_h`); MoE router / lm_head excluded - Quantization: Q4_K_M · File: `{GGUF_OUTFILE}` """ @app.function( image=eval_image, gpu="A100", timeout=60 * 60, secrets=[modal.Secret.from_name("huggingface")], volumes={"/root/.cache/huggingface": hf_cache}, ) def evaluate(repo: str, file: str, limit: int = 0) -> dict: """Run val prompts through a GGUF model and report JSON-parse / schema-pass rates. Runs on GPU in eval_image (llama.cpp), so it works even though the dev Codespace has no llama.cpp. Returns a metrics dict and logs a summary. """ import json import jsonschema from llama_cpp import Llama def extract_json(text: str): start = text.find("{") if start == -1: return None depth = 0 for i in range(start, len(text)): if text[i] == "{": depth += 1 elif text[i] == "}": depth -= 1 if depth == 0: raw = text[start:i + 1] if raw.startswith("{{") and raw.endswith("}}"): raw = raw[1:-1] return raw return None schema = json.loads(Path(REMOTE_SCHEMA).read_text()) rows = [json.loads(l) for l in Path(f"{REMOTE_DATA}/val.jsonl").read_text().splitlines() if l.strip()] if limit: rows = rows[:limit] print(f"[eval] {repo}/{file} — loading") llm = Llama.from_pretrained(repo_id=repo, filename=file, n_ctx=8192, n_gpu_layers=-1, verbose=False) parsed_ok = schema_ok = 0 for row in rows: messages = [m for m in row["messages"] if m["role"] in ("system", "user")] result = llm.create_chat_completion( messages=messages, max_tokens=8192, temperature=0.3, top_p=0.9, stop=["```"], ) text = result["choices"][0]["message"]["content"].strip() js = extract_json(text) if not js: continue try: game = json.loads(js) except json.JSONDecodeError: continue parsed_ok += 1 try: jsonschema.validate(instance=game, schema=schema) schema_ok += 1 except jsonschema.ValidationError: pass n = len(rows) metrics = { "model": f"{repo}/{file}", "n": n, "json_parse_rate": round(100 * parsed_ok / n, 1), "schema_pass_rate": round(100 * schema_ok / n, 1), } print(f"[eval] {metrics['model']}: parse={metrics['json_parse_rate']}% " f"schema={metrics['schema_pass_rate']}% (n={n})") return metrics @app.local_entrypoint() def main(eval_baseline: bool = True, eval_limit: int = 0): """Train + upload, then evaluate the fine-tuned model (and baseline) on the val set.""" train.remote() print("\n=== Evaluation ===") tuned = evaluate.remote(HF_REPO, GGUF_OUTFILE, eval_limit) print(f"Fine-tuned : schema {tuned['schema_pass_rate']}% parse {tuned['json_parse_rate']}%") if eval_baseline: base = evaluate.remote(BASELINE_REPO, BASELINE_FILE, eval_limit) print(f"Baseline : schema {base['schema_pass_rate']}% parse {base['json_parse_rate']}%") verdict = ("PASS — safe to swap" if tuned["schema_pass_rate"] >= base["schema_pass_rate"] else "REGRESSION — do not swap") print(f"Gate: fine-tuned >= baseline ? {verdict}") @app.local_entrypoint() def convert_only(): """Convert+upload from the committed merged model (no GPU, no peft). modal run --detach training/train_modal.py::convert_from_merged """ convert_from_merged.remote() @app.local_entrypoint() def eval_only(repo: str = HF_REPO, file: str = GGUF_OUTFILE, limit: int = 0): """Evaluate an already-uploaded GGUF without retraining. Example: modal run training/train_modal.py::eval_only modal run training/train_modal.py::eval_only --repo nvidia/... --file ....gguf """ print(evaluate.remote(repo, file, limit))