Download config.py from sathishphdai/system-admin-slm-1m: direct link, hf CLI and curl.
- Browser
- Download file 4.5 kB
-
https://huggingface.co/sathishphdai/system-admin-slm-1m/resolve/main/config.py
- Command line
-
hf download hf://sathishphdai/system-admin-slm-1m/config.py
-
curl -L -o config.py https://huggingface.co/sathishphdai/system-admin-slm-1m/resolve/main/config.py
4.5 kB
| #!/usr/bin/env python3 | |
| """ | |
| Configuration for System-Admin-SLM: A Role-Based SLM for System Admin. | |
| ~1B params, LLaMA-style architecture with RoPE β supports up to 1M token context. | |
| """ | |
| from dataclasses import dataclass, field | |
| from pathlib import Path | |
| from typing import Optional | |
| class SLMConfig: | |
| """All hyperparameters and paths in one place.""" | |
| # ββ Project paths ββββββββββββββββββββββββββββββββββββββββββββββ | |
| project_dir: Path = Path(__file__).resolve().parent | |
| data_dir: Path = field(default=None) | |
| tokenizer_dir: Path = field(default=None) | |
| checkpoint_dir: Path = field(default=None) | |
| # ββ Domain βββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| domain_name: str = "System Admin" | |
| domain_slug: str = "system_admin" | |
| tokenizer_filename: str = "system_admin_tokenizer.json" | |
| # ββ Tokenizer ββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| vocab_size: int = 32_768 | |
| min_frequency: int = 2 | |
| special_tokens: list = field( | |
| default_factory=lambda: [ | |
| "<pad>", "<unk>", "<bos>", "<eos>", | |
| "<|system|>", "<|user|>", "<|assistant|>", | |
| ] | |
| ) | |
| # ββ Model (~1B params, LLaMA-style with RoPE) βββββββββββββββββ | |
| n_layer: int = 32 | |
| n_head: int = 20 | |
| n_embd: int = 1600 | |
| block_size: int = 512 | |
| dropout: float = 0.05 | |
| bias: bool = False | |
| ffn_multiplier: float = 2.667 | |
| # ββ RoPE βββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| max_position_embeddings: int = 100_000_000_000 # 100B tokens via RoPE | |
| rope_theta: float = 50_000_000_000.0 # Scaled for 100B context | |
| # ββ Sliding Window βββββββββββββββββββββββββββββββββββββββββββββ | |
| sliding_window: Optional[int] = None | |
| # ββ Gradient Checkpointing (essential for 1B on 24GB) ββββββββββ | |
| gradient_checkpointing: bool = True | |
| # ββ Training βββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| batch_size: int = 1 | |
| gradient_accumulation_steps: int = 16 | |
| learning_rate: float = 2e-4 | |
| weight_decay: float = 0.1 | |
| max_epochs: int = 3 | |
| dataset_stride: int = 512 | |
| warmup_steps: int = 100 | |
| grad_clip: float = 1.0 | |
| eval_interval: int = 50 | |
| eval_samples: int = 10 | |
| log_interval: int = 10 | |
| device: str = "auto" | |
| # ββ Generation βββββββββββββββββββββββββββββββββββββββββββββββββ | |
| max_new_tokens: int = 1_000_000 # 1M output tokens | |
| temperature: float = 0.8 | |
| top_k: int = 50 | |
| top_p: float = 0.9 | |
| # ββ HuggingFace ββββββββββββββββββββββββββββββββββββββββββββββββ | |
| hf_repo_name: str = "system-admin-slm-1m" | |
| hf_model_card_tags: list = field(default_factory=lambda: ['sysadmin', 'linux', 'windows-server', 'networking', 'security', 'slm', 'llama-style', 'rope', '1m-context', 'from-scratch', '1b-params']) | |
| def __post_init__(self): | |
| if self.data_dir is None: | |
| self.data_dir = self.project_dir / "data" | |
| if self.tokenizer_dir is None: | |
| self.tokenizer_dir = self.project_dir / "tokenizer" | |
| if self.checkpoint_dir is None: | |
| self.checkpoint_dir = self.project_dir / "checkpoints" | |
| self.data_dir.mkdir(parents=True, exist_ok=True) | |
| self.tokenizer_dir.mkdir(parents=True, exist_ok=True) | |
| self.checkpoint_dir.mkdir(parents=True, exist_ok=True) | |
| if self.device == "auto": | |
| import torch | |
| if torch.cuda.is_available(): | |
| self.device = "cuda" | |
| elif hasattr(torch.backends, "mps") and torch.backends.mps.is_available(): | |
| self.device = "mps" | |
| else: | |
| self.device = "cpu" | |
| cfg = SLMConfig() | |