From 7ba410d7389e3da292aa71bfb236aa8233e37b8d Mon Sep 17 00:00:00 2001 From: Amit Bahree Date: Thu, 10 Sep 2026 04:11:51 -0700 Subject: [PATCH] MEAP review feedback - Ch1-5 --- ACCELERATORS.md | 37 ++--- README.md | 8 +- code/.env.example | 32 +++++ code/README.md | 14 +- code/chapter01/sidebar_outputs.json | 4 +- code/chapter02/README.md | 8 +- code/chapter02/quickstart.py | 11 +- code/chapter02/run_chapter5_adapter.py | 12 +- .../ch03_synthetic_data_generation.py | 56 +++++--- code/chapter05/README.md | 16 +-- .../examples/example_qlora_training_output.md | 2 +- .../scripts/listing_5_1_prepare_dataset.py | 4 +- code/chapter05/scripts/train_with_safety.py | 2 +- code/chapter05/train_lora.py | 4 +- code/chapter05/train_qlora.py | 4 +- code/common/env.py | 23 +++ code/common/gpu.py | 55 +++++++ code/data/README.md | 6 +- code/data/it_support_fmt/valid.jsonl | 50 +++++++ code/pyproject.toml | 134 +++++++++--------- code/scripts/reformat_it_answers.py | 60 +++++--- 21 files changed, 389 insertions(+), 153 deletions(-) create mode 100644 code/.env.example create mode 100644 code/common/gpu.py create mode 100644 code/data/it_support_fmt/valid.jsonl diff --git a/ACCELERATORS.md b/ACCELERATORS.md index 09057c2..abfa43f 100644 --- a/ACCELERATORS.md +++ b/ACCELERATORS.md @@ -23,9 +23,9 @@ chapters but is impractical for training. | 4 - ICL / RAG | ✓ | ✓ | ✓ | CPU-friendly; GPU optional | | 5 - LoRA | ✓ | ✓ | ✓ | | | 5 - QLoRA (4-bit) | ✓ | ✗ | ✓ | `bitsandbytes` 4-bit is CUDA/ROCm-only; on a Mac use the LoRA path | -| 6 - full-parameter SFT | ✓ | ✗ | ✓ | Needs ~24 GB; a 16 GB Mac runs out of memory (~18 GB) | -| 7 - distillation | ✓ | ✗ | ✓ | Hosts the chapter 6 teacher; same memory profile as chapter 6 | -| 8 - DPO | ✓ | ✗ | ✓ | Full-model preference optimisation; same memory profile | +| 6 - full-parameter SFT | ✓ | ✗ | ✓ | Needs ~32 GB in total (two 24 GB cards or one 40 GB card); a single 24 GB card and a 16 GB Mac both run out of memory | +| 7 - distillation | ✓ | ✗ | ✓ | The student is a LoRA adapter (fits one 12 GB card); the teacher-generation stage hosts the chapter 6 model for inference (~10 GB) | +| 8 - DPO | ✓ | ✗ | ✓ | Full-parameter DPO needs ~54 GB (three 24 GB cards or one 80 GB card); `--lora` fits one 24 GB card | | 9 - drift / registry / monitor | ✓ | ✓ | ✓ | Registry, drift detector, and rollback are CPU/stdlib; canary and safety monitor are inference | ✓ = validated to run. ✗ = does not run on that accelerator for *training* (use NVIDIA, AMD, or a cloud GPU) -- but the trained models for chapters 5 to 8 are published on Hugging Face, so you can still run their inference and evaluation on any machine (see [Running without training](#running-without-training-pull-the-model-from-hugging-face)). On Apple Silicon, training is correct but slower than on a GPU, so give it at least 16 GB of unified memory. @@ -34,12 +34,12 @@ chapters but is impractical for training. - **Any NVIDIA GPU with enough VRAM** runs everything; this is the reference path. - **An AMD GPU on Linux (ROCm)** also runs everything, including QLoRA and the full-parameter chapters. Best on datacenter (MI-series) cards; consumer RDNA support varies by GPU generation. -- **A Mac (Apple Silicon)** is great for chapters 1 through 5's LoRA path and chapter 9, but cannot *train* QLoRA or the full-parameter chapters (6, 7, 8). For training those, use a cloud GPU. +- **A Mac (Apple Silicon)** is great for chapters 1 through 5's LoRA path, chapter 7's LoRA student, and chapter 9, but cannot *train* QLoRA or the full-parameter chapters (6 and 8). For training those, use a cloud GPU. - **No GPU, or a Mac/small card?** You can still follow chapters 5 through 8 by pulling the trained model from Hugging Face and running inference or evaluation, without training it. See [Running without training](#running-without-training-pull-the-model-from-hugging-face). ## Running without training: pull the model from Hugging Face -You do not need a training-capable GPU to follow along. Every chapter's trained artifact is published to a single repo, [`bahree/ModelAdaptationBook`](https://huggingface.co/bahree/ModelAdaptationBook), as a per-chapter subfolder. Training a full-parameter model (chapters 6 to 8) needs a CUDA 24 GB+ card, but **loading the published model and running inference or evaluation fits a single smaller GPU or Apple Silicon (MPS)**. So on a Mac you can pull, for example, the chapter 6 SFT model and run its three-way evaluation without ever training it. +You do not need a training-capable GPU to follow along. Every chapter's trained artifact is published to a single repo, [`bahree/ModelAdaptationBook`](https://huggingface.co/bahree/ModelAdaptationBook), as a per-chapter subfolder. Training a full-parameter model (chapters 6 and 8) needs about 32 GB and 54 GB of GPU memory respectively (two or three 24 GB cards, or one 40 GB / 80 GB card), but **loading the published model and running inference or evaluation fits a single smaller GPU or Apple Silicon (MPS)**. So on a Mac you can pull, for example, the chapter 6 SFT model and run its three-way evaluation without ever training it. | Subfolder | Chapter | Artifact | Base | | --- | --- | --- | --- | @@ -55,18 +55,21 @@ Load a full model with `AutoModelForCausalLM.from_pretrained("bahree/ModelAdapta AMD (ROCm) was validated end-to-end on a datacenter card (Instinct MI300X); VRAM needs match the NVIDIA column, and consumer RDNA support varies by GPU generation. See the [AMD note](#amd-gpu-notes) below. -| Chapter | Minimum NVIDIA GPU | Recommended | AMD (ROCm) | CPU fallback | Apple Silicon (MPS) | -|---|---|---|---|---|---| -| 1 (sidebar reproducer) | None for base-only mode; 8 GB+ for the LoRA / SFT branches | 12 GB+ | Yes | Yes (base-only, slow) | Yes (base-only mode, slow on 8 GB unified memory) | -| 2 (LoRA quick-start) | 6 GB | 12 GB+ | Yes | Yes (slow) | Yes, verified on Apple M4/16 GB (quickstart trains on MPS in ~7 min) | -| 3 (data-quality experiment) | 8 GB | 12 GB+ | Yes | Synthetic-data pipeline yes; manifest module yes; full experiment slow | Yes, verified (bf16 LoRA on MPS); synthetic-data pipeline and manifest module also yes | -| 4 (ICL/RAG) | None for mock backends; ~8 GB for the optional Qwen3-4B HF backend | 12 GB+ | Yes | Yes (mock backend / hash embedder) | Yes (mock backends are CPU; HF backend uses MPS, slow on 8 GB) | -| 5 (LoRA) | 8 GB (RTX 3060/4060+) | 12 GB+ | Yes | Yes, but ~20× slower | Yes (trains on MPS) | -| 5 (QLoRA) | 6 GB | 8 GB+ | Yes (4-bit works; benign `rocminfo` warning) | Not recommended | **No** (`bitsandbytes` is CUDA-only) | -| 6 (Full SFT) | 24 GB (A30 / RTX 4090) | A100 40 GB+ | Yes | No | No | -| 7 (Distillation) | 12 GB (LoRA student) + 24 GB to host the chapter 6 teacher | 24 GB+ | Yes | Not recommended | No | -| 8 (DPO) | 24 GB | A100 40 GB+ | Yes | No | No | -| 9 (Drift / Registry / Monitor) | None for the CPU stages (registry, drift detector, rollback demo); ~8 GB for the GPU stages (canary, safety monitor) | 12 GB+ | Yes | Yes for stages 1, 2, and 4 | Yes for stages 1, 2, and 4 | +| Chapter | Measured peak (A30, repo defaults) | Minimum NVIDIA GPU | Recommended | AMD (ROCm) | CPU fallback | Apple Silicon (MPS) | +|---|---|---|---|---|---|---| +| 1 (sidebar reproducer) | inference only (~8 GB to load the 4B model in bf16) | None for base-only mode; 8 GB+ for the LoRA / SFT branches | 12 GB+ | Yes | Yes (base-only, slow) | Yes (base-only mode, slow on 8 GB unified memory) | +| 2 (LoRA quick-start) | **9.0 GB** allocated (10.9 GB reserved), one card; 40 examples, 20 steps, 77 s on an A30 | 6 GB | 12 GB+ | Yes | Yes (slow) | Yes, verified on Apple M4/16 GB (quickstart trains on MPS in ~7 min) | +| 3 (data-quality experiment) | same profile as chapter 5 LoRA | 8 GB | 12 GB+ | Yes | Synthetic-data pipeline yes; manifest module yes; full experiment slow | Yes, verified (bf16 LoRA on MPS); synthetic-data pipeline and manifest module also yes | +| 4 (ICL/RAG) | inference only (~8 GB for the optional HF backend) | None for mock backends; ~8 GB for the optional Qwen3-4B HF backend | 12 GB+ | Yes | Yes (mock backend / hash embedder) | Yes (mock backends are CPU; HF backend uses MPS, slow on 8 GB) | +| 5 (LoRA, r=16) | **9.0 GB** allocated (10.2 GB reserved), one card | 8 GB (RTX 3060/4060+) | 12 GB+ | Yes | Yes, but ~20× slower | Yes (trains on MPS) | +| 5 (QLoRA, r=8) | **5.1 GB** allocated (5.4 GB reserved), one card | 6 GB | 8 GB+ | Yes (4-bit works; benign `rocminfo` warning) | Not recommended | **No** (`bitsandbytes` is CUDA-only) | +| 6 (Full SFT) | **32.5 GB** total (17.2 + 15.3 GB across two A30s); **OOM on one A30** at 23.2 GB during AdamW state init | Two 24 GB cards (A30 / RTX 4090), or one 40 GB card | A100 40 GB+ | Yes | No | No | +| 7 (Distillation, LoRA student r=16) | **10.5 GB** allocated (15.8 GB reserved, batch 2), one card | 12 GB for the student; the teacher-generation stage loads the chapter 6 model for inference (~10 GB) | 24 GB+ | Yes | Not recommended | Student: yes in principle (LoRA on MPS); teacher generation is slow | +| 8 (DPO, full-parameter) | **54.3 GB** total (18.4 + 18.6 + 17.3 GB across three A30s); **OOM on two A30s** (23.2 + 22.7 GB) | Three 24 GB cards, or one 80 GB card | A100 80 GB / H100 | Yes | No | No | +| 8 (DPO, `--lora`, r=16) | **10.6 GB** allocated (14.7 GB reserved), one card | 12 GB | 24 GB | Yes | No | No (`ch8-dpo-lora` inference only) | +| 9 (Drift / Registry / Monitor) | inference only (~8 GB for the GPU stages) | None for the CPU stages (registry, drift detector, rollback demo); ~8 GB for the GPU stages (canary, safety monitor) | 12 GB+ | Yes | Yes for stages 1, 2, and 4 | Yes for stages 1, 2, and 4 | + +**How the measured column was produced (2026-09-10).** Each training script was run for three steps with its defaults (Qwen3-4B-Instruct-2507, `max_length` 512, bf16, gradient checkpointing) on NVIDIA A30 24 GB cards with PyTorch 2.11+cu126, and `torch.cuda.max_memory_allocated()` read per device at the end; "total" sums the devices when `device_map="auto"` sharded the model, which is the number a single card would need. Multi-card totals came from the same box that produced the book's published runs, which is why the chapter 6 and chapter 8 text and earlier versions of this table understated the single-card requirement. Every training script now prints its own peak when training ends, and you can re-measure any chapter on your hardware with `python -m scripts.measure_peak_vram --max_steps 3` (see `code/scripts/measure_peak_vram.py`). Why the fixed floors: full SFT keeps bf16 weights, bf16 gradients, and two bf16 AdamW moments, 8 bytes per parameter, about 32 GB for 4B parameters before activations; full DPO adds a frozen bf16 reference copy (another 8 GB) and two forward passes per preference pair. Gradient checkpointing trims activations, not these floors. An 8-bit optimizer would bring full SFT to roughly 24 GB plus activations, still not a comfortable single-A30 fit; CPU optimizer offload (not used in the book's scripts) is the only single-24 GB-card route. **Disk space:** budget about 50 GB free for the Hugging Face model cache plus chapter 6's run directory (full-parameter checkpoints with optimizer state are 22-24 GB each). See `code/chapter06/README.md` for the breakdown. diff --git a/README.md b/README.md index ea5d576..80b7e6b 100644 --- a/README.md +++ b/README.md @@ -60,7 +60,7 @@ You need working Python and comfort at the command line; the book teaches the LL - **The vendor-neutral, do-it-yourself counterpart to managed fine-tuning.** As the major vendors turn enterprise fine-tuning into a managed service, this book teaches the underlying methods and the operations layer those services abstract away, so you can run them on your own models and your own hardware and weigh build versus buy with real numbers. - **The full technique spectrum in one running example.** A single fictitious enterprise (Contoso) and its IT help desk is threaded from prompting all the way through alignment and operations, so you see the same problem solved at every rung, with an explicit cost, latency, privacy, and ROI framework for choosing between the techniques. -- **Runnable and reproducible on a single GPU.** Every chapter runs on one GPU (LoRA and QLoRA on a modest consumer card; full fine-tuning and DPO on a single 24 GB card) and reproduces the book's published numbers within run-to-run variance. Validated on NVIDIA, AMD, and Apple Silicon. +- **Runnable and reproducible without a cluster.** LoRA, QLoRA, and the distilled student run on one modest consumer card; full fine-tuning and full-parameter DPO of the 4B model need about 32 GB, so one 40 GB card or two 24 GB cards (the book's runs used two A30s). Every chapter reproduces the book's published numbers within run-to-run variance. Validated on NVIDIA, AMD, and Apple Silicon. - **Honest engineering, in the open.** The book shows when a technique does *not* win (for example, where DPO matches SFT on objective accuracy), anchors claims to real cost economics, and runs a safety-regression check at each step. The code, the trained models, and the training and evaluation logs are all public, including the experiments that did not work, so you can verify every result and learn as much from what failed as from what worked. - **The operational layer most LLM books skip.** Drift detection, versioning, rollback, and continuous safety monitoring, the part that decides whether a fine-tuned model survives past launch. @@ -71,7 +71,7 @@ It uses a small open-weights model (Qwen3-4B) so every result reproduces on mode **The model: `Qwen/Qwen3-4B-Instruct-2507`.** One open-weights model is the spine of every chapter, chosen so the choices are realistic and the results reproduce on accessible hardware: - **Open-weights.** You own, host, inspect, and fine-tune it; nothing depends on a vendor API. That is the book's whole stance. -- **4B fits a single GPU.** LoRA and QLoRA train on a modest consumer card; full SFT and DPO fit a single 24 GB card. Every chapter reproduces without a cluster. +- **4B fits small hardware.** LoRA and QLoRA train on a modest consumer card (9 GB and 5 GB measured); full SFT and full DPO need about 32 GB, which is two 24 GB cards or one 40 GB card, not a cluster. LoRA-DPO fits one 24 GB card. - **Already instruction-tuned.** A realistic enterprise starting point, so each technique's effect is meaningful rather than teaching basic instruction-following. - **One consistent base across all chapters.** The chapter 5 LoRA, chapter 6 SFT, chapter 7 distilled student, and chapter 8 DPO model all build on the same spine, so the running example chains and the comparisons stay apples-to-apples. - **Permissive license and strong quality-for-size**, and the techniques are model-agnostic, so they apply unchanged to larger frontier models. @@ -161,6 +161,8 @@ python -m pip install -U pip pip install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cpu ``` +**Python version.** The code is validated on Python 3.12 (also what CI runs). Newer releases work once PyTorch publishes wheels for them, but those wheels lag a new Python by weeks to months, and a reader on Python 3.14 hit `ERROR: No matching distribution found for torch` from the command above. If you see that error, create the virtual environment with Python 3.12 (`python3.12 -m venv .venv`) and rerun the install. + For other CUDA versions (12.1, 11.8) or to confirm the right command for your machine, see the official selector at . `code/README.md` has more detail, including NVIDIA driver install steps for fresh Ubuntu/Proxmox VMs. Not sure which accelerator runs which chapter, or how much GPU memory you need? See **[ACCELERATORS.md](ACCELERATORS.md)**. **3. Install the book package and smoke-test:** @@ -183,7 +185,7 @@ The full book runs on NVIDIA (CUDA) and AMD (ROCm) GPUs; most of it also runs on See also **[LESSONS.md](LESSONS.md)** for the reusable, hard-won gotchas behind these results: pin the model to a device rather than relying on `device_map="auto"`, Hugging Face rate limits on datacenter IPs, and the Apple Silicon, AMD ROCm, and Blackwell notes. -Two common gotchas, both covered there: chapter 5's QLoRA needs an NVIDIA or AMD GPU ([why](ACCELERATORS.md#why-qlora-needs-an-nvidia-or-amd-gpu)), and the full-parameter chapters (6, 7, 8) need ~24 GB so they do not fit a 16 GB Mac. +Two common gotchas, both covered there: chapter 5's QLoRA needs an NVIDIA or AMD GPU ([why](ACCELERATORS.md#why-qlora-needs-an-nvidia-or-amd-gpu)), and the full-parameter chapters (6 and 8) need about 32 GB of GPU memory (two 24 GB cards or one 40 GB card), so they do not fit a 16 GB Mac or a single 24 GB card; chapter 7's student is LoRA and fits one card. ## Support diff --git a/code/.env.example b/code/.env.example new file mode 100644 index 0000000..2c61b3c --- /dev/null +++ b/code/.env.example @@ -0,0 +1,32 @@ +# code/.env.example — copy to code/.env and fill in the keys you need. +# +# cp code/.env.example code/.env +# +# code/.env is gitignored. It is loaded automatically by common/env.py +# (python-dotenv) for any code run from the code/ directory. Every key below is +# OPTIONAL — you only need the ones for the chapters/features you run. + +# --- OpenRouter: frontier-model API calls --- +# Used by chapter 8 concise-preference build + judge/eval, the contoso_qa_demo, +# and common/openrouter.py. Get a key at https://openrouter.ai/keys +OPENROUTER_API_KEY= + +# --- Anthropic: chapter 3 synthetic-data generation (teacher = Claude) --- +# Get a key at https://console.anthropic.com/ (or set TEACHER_BACKEND to another backend) +ANTHROPIC_API_KEY= + +# --- Hugging Face token (optional) --- +# Speeds up model/dataset downloads; REQUIRED only to publish adapters/models +# (chapter 5 publish scripts). Create at https://huggingface.co/settings/tokens +HF_TOKEN= + +# --- Weights & Biases experiment tracking (off by default) --- +# Leave WANDB_DISABLED=1 to keep tracking off. To enable: set it to 0 and run +# `wandb login` (or set WANDB_API_KEY), plus an optional project name. +WANDB_DISABLED=1 +# WANDB_API_KEY= +# WANDB_PROJECT=amitbahree/huggingface + +# --- Misc toggles used by validation/tests (optional) --- +# BOOKCODE_SKIP_MODEL_DOWNLOAD=1 # skip the model download/inference smoke step +# BOOKCODE_REPORT_TO=none # trainer report_to override diff --git a/code/README.md b/code/README.md index f4b6b57..a337dd3 100644 --- a/code/README.md +++ b/code/README.md @@ -51,13 +51,15 @@ python3 --version **If your version is below 3.12**, install a newer Python version before proceeding. +**If your version is the newest Python release** (for example 3.14 in the months after it shipped) and the PyTorch install step later fails with `No matching distribution found for torch`, PyTorch has not published wheels for that interpreter yet. Use Python 3.12, which is what the book's code is validated on: `python3.12 -m venv .venv`. + ### Ubuntu/Debian Prerequisites **If you're on Ubuntu/Debian**, install the `venv` package first: ```bash sudo apt update -sudo apt install python3.12-venv # or python3.10-venv, python3.11-venv depending on your version +sudo apt install python3.12-venv # match your Python 3.12+ version, e.g. python3.13-venv ``` ### Create Virtual Environment @@ -262,6 +264,12 @@ echo 'export HF_TOKEN="hf_..."' >> ~/.bashrc **Note:** If you don't set a token, downloads still work but you may see warnings about unauthenticated requests. This is harmless. +### Download troubleshooting + +- **`Fast download using 'hf_transfer' is enabled but 'hf_transfer' package is not available`**: your environment exports `HF_HUB_ENABLE_HF_TRANSFER=1` (some cloud images and notebooks do) without the package. The book's scripts detect this and fall back to the standard downloader automatically; if you call the Hugging Face libraries yourself, either `pip install hf_transfer` or `unset HF_HUB_ENABLE_HF_TRANSFER`. Nothing in this repo needs hf_transfer. +- **`An error occurred while downloading using hf_transfer`** (usually behind a proxy or on a flaky connection): `unset HF_HUB_ENABLE_HF_TRANSFER` and retry; the standard downloader resumes partial files. +- **Slow first download**: the base model is about 8 GB. Set `HF_TOKEN` (above) to lift the anonymous rate limit; the download is cached in `~/.cache/huggingface` and never repeated. + ## API keys (optional) A few scripts call a hosted LLM API. **None of the core Chapter 1-5 hands-on work (LoRA/QLoRA training, evaluation, inference) needs an API key**, and the IT support dataset is already committed under `data/it_support*`, so you only need a key if you want to *rebuild the data from scratch* or run the optional generation / judging scripts. @@ -321,8 +329,8 @@ The hands-on chapters train on the book's **IT-support dataset**: real Stack Exc # 1. Build the dataset -> data/it_support/ (train.jsonl, valid.jsonl, preferences.jsonl, manifest.json, attribution.jsonl) python scripts/build_it_support_dataset.py -# 2. Reformat the answers into the house style -> data/it_support_fmt/train.jsonl -python scripts/reformat_it_answers.py +# 2. Reformat the answers into the house style -> data/it_support_fmt/{train,valid}.jsonl +python scripts/reformat_it_answers.py # (needs OPENROUTER_API_KEY; the output file is already committed, so this step is optional unless you rebuild from source) ``` The builder depends on `beautifulsoup4` (to clean the Stack Exchange HTML) and `datasets`, both pulled in by the base install. Per-example source URLs for the Stack Exchange content are written to `data/it_support/attribution.jsonl` (dataset licenses and redistribution terms are in the [main README](../README.md#license-and-data-attribution)). The chapter 2 quickstart and chapter 5 onward load the prepared data from `data/it_support_fmt` (and `data/it_support/valid.jsonl`). diff --git a/code/chapter01/sidebar_outputs.json b/code/chapter01/sidebar_outputs.json index f4e9592..2b6f983 100644 --- a/code/chapter01/sidebar_outputs.json +++ b/code/chapter01/sidebar_outputs.json @@ -4,6 +4,6 @@ "model": "Qwen/Qwen3-4B-Instruct-2507", "decoding": "greedy", "base": "If you've forgotten your Outlook password, here\u2019s what you should do \u2014 depending on whether you're using **Outlook on a desktop**, **Outlook Web App (OWA)**, or **a corporate email system**:\n\n---\n\n### \ud83d\udd10 1. **Try to Reset Your Password via the Email Provider**\nMost corporate or organization-managed Outlook accounts (like Microsoft 365) allow password reset through the email login page.\n\n#### Steps:\n1. Go to [https://outlook.com](https://outlook.com) or your organization's email portal.\n2. Click on **\"Forgot password?\"** (usually at the bottom of the login screen).\n3. Enter your **email address** associated with the account.\n4. Follow the instructions sent to your email (usually a link to reset the password).\n5. Create a new, secure password and confirm it.\n\n> \u26a0\ufe0f Note: If you don\u2019t receive the email, check your spam folder or contact your IT department.\n\n---\n\n### \ud83d\udda5\ufe0f 2. **If You\u2019re Using Outlook Desktop (e.g., Outlook for Windows/Mac)**\n- The password is tied to your email account.\n- If you can't log in, you\u2019ll need to reset it via the web or through", - "lora": "If you have forgotten your Outlook password, you can reset it by clicking on the \"Forgot Password\" link on the login page. You will then be prompted to enter your email address. An email with a link to reset your password will be sent to that address.", - "sft": "If you have forgotten your Outlook password, you can reset it using the following steps:\n\n1. Open the Outlook application on your computer.\n2. Click on the \"File\" tab in the top-left corner of the Outlook window.\n3. In the File menu, click on the \"Account Settings\" option.\n4. In the Account Settings window, click on the \"Change Password\" button.\n5. In the Change Password window, enter your current password in the \"Current Password\" field.\n6. Enter your new password in the \"New Password\" field.\n7. Confirm your new password by entering it again in the \"Confirm New Password\" field.\n8. Click on the \"OK\" button to save your new password.\n9. Click on the \"Close\" button to close the Account Settings window.\n\nAfter completing these steps, you will have successfully reset your Outlook password." + "lora": "**Summary:** Reset your Outlook password through the Microsoft Online Services Recovery Tool or by contacting your administrator.\n\n**Steps:**\n1. Use the Microsoft Online Services Recovery Tool to reset your password.\n2. If you cannot use the tool, contact your administrator for assistance.", + "sft": "**Summary:** Reset your Outlook password using the password reset form available at https://helpdesk.somewhere.com.\n\n**Steps:**\n1. Visit the password reset form at https://helpdesk.somewhere.com.\n2. Enter your email address to initiate the password reset process." } \ No newline at end of file diff --git a/code/chapter02/README.md b/code/chapter02/README.md index 27abcc5..1e51852 100644 --- a/code/chapter02/README.md +++ b/code/chapter02/README.md @@ -38,8 +38,8 @@ The quickstart trains on the book's **IT-support dataset**: real Stack Exchange # 1. Build the dataset -> data/it_support/ (train.jsonl, valid.jsonl, preferences.jsonl, manifest.json, attribution.jsonl) python scripts/build_it_support_dataset.py -# 2. Reformat answers into the house style -> data/it_support_fmt/train.jsonl -python scripts/reformat_it_answers.py +# 2. Reformat answers into the house style -> data/it_support_fmt/{train,valid}.jsonl +python scripts/reformat_it_answers.py # (needs OPENROUTER_API_KEY; the output file is already committed, so this step is optional unless you rebuild from source) ``` The builder depends on `beautifulsoup4` (to clean the Stack Exchange HTML) and `datasets`, both part of the base install. Per-example source URLs for the Stack Exchange content are written to `data/it_support/attribution.jsonl`. The quickstart then loads `data/it_support_fmt/train.jsonl` and `data/it_support/valid.jsonl`. @@ -57,7 +57,7 @@ pip install -e ".[dev]" # Build the IT-support dataset once (see "Preparing the data" above) python scripts/build_it_support_dataset.py -python scripts/reformat_it_answers.py +python scripts/reformat_it_answers.py # (needs OPENROUTER_API_KEY; the output file is already committed, so this step is optional unless you rebuild from source) # Run the five-step LoRA fine-tune python -m chapter02.quickstart @@ -70,7 +70,7 @@ Step 1: prepare dataset train=40 valid=5 demo=3 Step 2: load base model and configure LoRA Step 3: train for 20 steps - train wall time: ~56s on A30 + train wall time: ~78s on A30 (peak GPU memory about 9 GB) Step 4: compare outputs on held-out prompts ... Step 5: save adapter to chapter02/runs/ch2_quickstart diff --git a/code/chapter02/quickstart.py b/code/chapter02/quickstart.py index 424cf61..1bbf263 100644 --- a/code/chapter02/quickstart.py +++ b/code/chapter02/quickstart.py @@ -37,6 +37,9 @@ import torch from datasets import Dataset as HFDataset + +import common.env # noqa: F401 loads code/.env and guards Hugging Face download settings +from common.gpu import report_peak_gpu_memory from peft import LoraConfig from transformers import AutoModelForCausalLM, AutoTokenizer from trl import SFTConfig, SFTTrainer @@ -96,8 +99,12 @@ def step1_prepare_dataset() -> tuple[HFDataset, HFDataset, List[Dict[str, Any]]] """ print("Step 1: prepare dataset") from common.jsonl import read_jsonl + # Both files carry answers in the house format (built by scripts/build_it_support_dataset.py, + # then scripts/reformat_it_answers.py), so training loss and eval loss score the same style. + # The raw answers in data/it_support/valid.jsonl stay the held-out test set the later chapters + # evaluate against; the prompts are identical in both files. trows = list(read_jsonl("data/it_support_fmt/train.jsonl")) - vrows = list(read_jsonl("data/it_support/valid.jsonl")) + vrows = list(read_jsonl("data/it_support_fmt/valid.jsonl")) def role_of(row: Dict[str, Any], role: str) -> str: return next(m["content"] for m in row["messages"] if m["role"] == role) @@ -197,6 +204,7 @@ def step3_train(model, tokenizer, lora_config, train_ds, valid_ds) -> SFTTrainer output_dir=str(OUTPUT_DIR), max_steps=MAX_STEPS, per_device_train_batch_size=1, + per_device_eval_batch_size=1, # match chapter 5; the TRL default of 8 adds ~3 GB at eval time gradient_accumulation_steps=8, learning_rate=2e-4, warmup_ratio=0.1, @@ -222,6 +230,7 @@ def step3_train(model, tokenizer, lora_config, train_ds, valid_ds) -> SFTTrainer t0 = time.time() trainer.train() print(f" train wall time: {time.time() - t0:.1f}s") + report_peak_gpu_memory("the 20 quickstart steps") return trainer diff --git a/code/chapter02/run_chapter5_adapter.py b/code/chapter02/run_chapter5_adapter.py index 3acc8fb..2113c84 100644 --- a/code/chapter02/run_chapter5_adapter.py +++ b/code/chapter02/run_chapter5_adapter.py @@ -11,7 +11,8 @@ Adapter resolution order (first match wins): 1. Local copy at chapter05/runs/it_lora/ (if you have already run Ch5) - 2. Hugging Face Hub at the published location (cached after first use) + 2. Hugging Face Hub at the published location (public and ungated, so no HF_TOKEN is + needed; downloaded once into ~/.cache/huggingface and read from disk after that) 3. Local chapter 2 quickstart adapter at chapter02/runs/ch2_quickstart/ (only when --use-quickstart is passed; the quickstart adapter is NOT a chapter 5 adapter, the script will say so) @@ -34,6 +35,8 @@ from peft import PeftModel from transformers import AutoModelForCausalLM, AutoTokenizer +import common.env # noqa: F401 loads code/.env and guards Hugging Face download settings + BASE_MODEL = "Qwen/Qwen3-4B-Instruct-2507" CH5_LOCAL_PATH = Path("chapter05/runs/it_lora") @@ -114,11 +117,12 @@ def print_no_adapter_instructions(args: argparse.Namespace) -> None: print(" python scripts/reformat_it_answers.py") print(" python -m chapter05.train_lora \\") print(" --train data/it_support_fmt/train.jsonl \\") - print(" --valid data/it_support/valid.jsonl \\") + print(" --valid data/it_support_fmt/valid.jsonl \\") print(f" --out {CH5_LOCAL_PATH}") print() - print("Option B. Pull the published adapter from Hugging Face Hub once it is up:") - print(f" (the script tries {args.hub_repo} automatically when it is published).") + print("Option B. Pull the published adapter from Hugging Face Hub (public, no token needed):") + print(f" (the script tries {args.hub_repo} automatically; this message means the Hub was unreachable,") + print(" so check your network or proxy, or pass --hub-repo bahree/ModelAdaptationBook#ch5-lora).") print() print("Option C. Run the chapter 2 quickstart and pass --use-quickstart:") print(" python -m chapter02.quickstart") diff --git a/code/chapter03/ch03_synthetic_data_generation.py b/code/chapter03/ch03_synthetic_data_generation.py index f94afc2..07b04bc 100644 --- a/code/chapter03/ch03_synthetic_data_generation.py +++ b/code/chapter03/ch03_synthetic_data_generation.py @@ -549,10 +549,13 @@ def check_distribution_alignment( # narrows and the model underperforms on low-frequency inputs — exactly the # rare-but-important examples that fine-tuning is supposed to improve. # -# The formula: n_synth = n_real × 0.30 / (1 - 0.30) -# = n_real × 0.4286 -# Solving for n_synth given n_real ensures the synthetic proportion stays -# at exactly 30% after mixing, regardless of how many passed the quality gate. +# The formula: n_synth = n_train_real × 0.30 / (1 - 0.30) +# = n_train_real × 0.4286 +# Solving for n_synth given the real count ensures the synthetic proportion +# stays at exactly 30% after mixing, regardless of how many passed the gate. +# Note the input is the real count *in the train split*, after the eval +# examples have been held out; synthetic data never enters the eval split, +# so sizing the cap against the full real pool would overshoot 30%. # # The manifest records the SHA-256 hash of the full mixed dataset. Any future # change to the data — even a single character — will produce a different hash, @@ -572,6 +575,10 @@ def mix_and_save( This ensures the evaluation benchmark measures against ground truth, not against the teacher model's preferred outputs. + The eval examples are held out FIRST and never enter the train split. + Train and eval must stay strictly disjoint, otherwise the eval score + measures memorization rather than generalization. + Args: real_examples: Real seed examples (the full available pool) synthetic_examples: Verified synthetic examples from Step 4 @@ -585,12 +592,22 @@ def mix_and_save( import os os.makedirs(output_dir, exist_ok=True) - n_real = len(real_examples) - # Cap synthetic count to enforce the 30% ratio - n_synth_max = int(n_real * max_synth_ratio / (1 - max_synth_ratio)) + rng = random.Random(seed) + n_real = len(real_examples) + + # Hold the eval examples out FIRST, before anything is mixed. Everything + # that lands in eval is removed from the real pool, so no eval example + # can reach the train split. + n_eval = min(max(5, n_real // 10), n_real) + real_shuffled = list(real_examples) + rng.shuffle(real_shuffled) + eval_real = real_shuffled[:n_eval] + train_real = real_shuffled[n_eval:] + + # Cap synthetic count to enforce the 30% ratio inside the train split, + # the only split synthetic data is allowed into. + n_synth_max = int(len(train_real) * max_synth_ratio / (1 - max_synth_ratio)) n_synth = min(n_synth_max, len(synthetic_examples)) - - rng = random.Random(seed) synth_sample = rng.sample(synthetic_examples, n_synth) # Convert to ChatML format for training @@ -608,14 +625,15 @@ def to_chatml(ex: dict) -> dict: "label": ex["label"], } - all_rows = [to_chatml(ex) for ex in real_examples] + \ - [to_chatml(ex) for ex in synth_sample] - rng.shuffle(all_rows) + # Train gets the remaining real examples plus the capped synthetic sample. + # Eval gets the held-out real examples only. The two are disjoint by + # construction, because eval_real was removed from the pool above. + train_rows = [to_chatml(ex) for ex in train_real] + \ + [to_chatml(ex) for ex in synth_sample] + eval_rows = [to_chatml(ex) for ex in eval_real] + rng.shuffle(train_rows) - # 90/10 train/eval split — eval uses only real examples - n_eval = max(5, len(real_examples) // 10) - eval_rows = [to_chatml(ex) for ex in rng.sample(real_examples, n_eval)] - train_rows = all_rows + all_rows = train_rows + eval_rows # full dataset, for the manifest hash # Save as HuggingFace Datasets Dataset.from_list(train_rows).save_to_disk(f"{output_dir}/train") @@ -635,7 +653,7 @@ def to_chatml(ex: dict) -> dict: "eval": len(eval_rows), "real": n_real, "synthetic": n_synth, - "synth_pct": round(n_synth / len(all_rows), 3), + "synth_pct": round(n_synth / max(len(train_rows), 1), 3), }, "synthetic_cap": max_synth_ratio, "trained_models": [], # populated after training with checkpoint info @@ -646,9 +664,9 @@ def to_chatml(ex: dict) -> dict: print(f"\nStep 6: Mixed dataset saved to {output_dir}/") print(f" Real : {n_real}") - print(f" Synthetic : {n_synth} ({manifest['composition']['synth_pct']:.0%} of total)") + print(f" Synthetic : {n_synth} ({manifest['composition']['synth_pct']:.0%} of train)") print(f" Train : {len(train_rows)}") - print(f" Eval : {len(eval_rows)} (real only)") + print(f" Eval : {len(eval_rows)} (real only, held out of train)") print(f" SHA-256 : {sha256[:24]}...") return manifest diff --git a/code/chapter05/README.md b/code/chapter05/README.md index 1857c7c..b451ca4 100644 --- a/code/chapter05/README.md +++ b/code/chapter05/README.md @@ -148,13 +148,13 @@ Build the IT support dataset, then reformat the answers into the house style. Ru ```bash # From code/chapter05/ (venv active) python scripts/build_it_support_dataset.py -python scripts/reformat_it_answers.py +python scripts/reformat_it_answers.py # (needs OPENROUTER_API_KEY; the output file is already committed, so this step is optional unless you rebuild from source) ``` **Windows (PowerShell/CMD):** ```powershell python scripts\build_it_support_dataset.py -python scripts\reformat_it_answers.py +python scripts\reformat_it_answers.py # (needs OPENROUTER_API_KEY; the output file is already committed, so this step is optional unless you rebuild from source) ``` This will: @@ -164,7 +164,7 @@ This will: - Write a `manifest.json` and an `attribution.jsonl` recording the per-example source URL and license - Save to `data/it_support/` (`train.jsonl`, `valid.jsonl`, `preferences.jsonl`, `manifest.json`, `attribution.jsonl`) -The second script (`reformat_it_answers.py`) rewrites the answers into the chapter's house format and writes `data/it_support_fmt/train.jsonl`, the file you train on. +The second script (`reformat_it_answers.py`) rewrites the answers into the chapter's house format and writes `data/it_support_fmt/train.jsonl` (the file you train on) and `data/it_support_fmt/valid.jsonl` (the training-time validation split, processed the same way so eval loss is comparable to training loss). The raw `data/it_support/valid.jsonl` stays the held-out test set for the evaluation scripts, so token-F1 is still scored against the original human answers. **Resulting files:** ``` @@ -190,7 +190,7 @@ Train a LoRA adapter using TRL's SFTTrainer: ```bash python -m chapter05.train_lora \ --train data/it_support_fmt/train.jsonl \ - --valid data/it_support/valid.jsonl \ + --valid data/it_support_fmt/valid.jsonl \ --out chapter05/runs/it_lora ``` @@ -198,7 +198,7 @@ python -m chapter05.train_lora \ ```powershell python -m chapter05.train_lora ^ --train data/it_support_fmt/train.jsonl ^ - --valid data/it_support/valid.jsonl ^ + --valid data/it_support_fmt/valid.jsonl ^ --out chapter05/runs/it_lora ``` @@ -278,7 +278,7 @@ QLoRA uses 4-bit quantization, enabling training on smaller GPUs. (You already i ```bash python -m chapter05.train_qlora \ --train data/it_support_fmt/train.jsonl \ - --valid data/it_support/valid.jsonl \ + --valid data/it_support_fmt/valid.jsonl \ --out chapter05/runs/it_qlora ``` @@ -286,7 +286,7 @@ python -m chapter05.train_qlora \ ```powershell python -m chapter05.train_qlora ^ --train data/it_support_fmt/train.jsonl ^ - --valid data/it_support/valid.jsonl ^ + --valid data/it_support_fmt/valid.jsonl ^ --out chapter05/runs/it_qlora ``` @@ -409,7 +409,7 @@ The evaluation also runs a safety suite to ensure fine-tuning didn't weaken safe ### **"Dataset not found"** - **Run `build_it_support_dataset.py` then `reformat_it_answers.py` first** (Step 1) -- Check that files exist: `data/it_support_fmt/train.jsonl` and `data/it_support/valid.jsonl` +- Check that files exist: `data/it_support_fmt/train.jsonl`, `data/it_support_fmt/valid.jsonl` (training), and `data/it_support/valid.jsonl` (evaluation) ### "TRL not installed" - Install: `pip install trl>=0.9.0` diff --git a/code/chapter05/examples/example_qlora_training_output.md b/code/chapter05/examples/example_qlora_training_output.md index 70dae02..e8162d1 100644 --- a/code/chapter05/examples/example_qlora_training_output.md +++ b/code/chapter05/examples/example_qlora_training_output.md @@ -7,7 +7,7 @@ This file captures a typical run of `train_qlora` for the Chapter 5 IT support d ```bash python -m chapter05.train_qlora \ --train data/it_support_fmt/train.jsonl \ - --valid data/it_support/valid.jsonl \ + --valid data/it_support_fmt/valid.jsonl \ --out chapter05/runs/it_qlora ``` diff --git a/code/chapter05/scripts/listing_5_1_prepare_dataset.py b/code/chapter05/scripts/listing_5_1_prepare_dataset.py index 3e611d5..67db53a 100644 --- a/code/chapter05/scripts/listing_5_1_prepare_dataset.py +++ b/code/chapter05/scripts/listing_5_1_prepare_dataset.py @@ -10,7 +10,7 @@ Then train/evaluate against: --train data/it_support_fmt/train.jsonl - --valid data/it_support/valid.jsonl + --valid data/it_support_fmt/valid.jsonl This module is retained only because earlier drafts referenced it and the ``dolly_to_messages`` helper below documents the original Dolly format. Running @@ -92,7 +92,7 @@ def main() -> None: print() print("Then train and evaluate against:") print(" --train data/it_support_fmt/train.jsonl") - print(" --valid data/it_support/valid.jsonl") + print(" --valid data/it_support_fmt/valid.jsonl") if __name__ == "__main__": diff --git a/code/chapter05/scripts/train_with_safety.py b/code/chapter05/scripts/train_with_safety.py index 4504385..7cd4994 100644 --- a/code/chapter05/scripts/train_with_safety.py +++ b/code/chapter05/scripts/train_with_safety.py @@ -141,7 +141,7 @@ def main(): print("\nCommand:") print(" python -m chapter05.train_lora \\") print(f" --train {mixed_path} \\") - print(" --valid data/it_support/valid.jsonl \\") + print(" --valid data/it_support_fmt/valid.jsonl \\") print(f" --out {out_dir / 'adapter'} \\") print(" --epochs 3") diff --git a/code/chapter05/train_lora.py b/code/chapter05/train_lora.py index 48a4845..3b6a063 100644 --- a/code/chapter05/train_lora.py +++ b/code/chapter05/train_lora.py @@ -6,7 +6,7 @@ Usage: python -m chapter05.train_lora \\ --train data/it_support_fmt/train.jsonl \\ - --valid data/it_support/valid.jsonl \\ + --valid data/it_support_fmt/valid.jsonl \\ --out chapter05/runs/it_lora See Chapter 5, Section 5.1 (Step 2) and the README for full details. @@ -26,6 +26,7 @@ from chapter05.dataset import prepare_dataset_for_sft from chapter05.modeling import create_lora_config, load_base_model_lora, load_tokenizer from common.env import resolve_report_to +from common.gpu import report_peak_gpu_memory from common.seed import seed_everything @@ -149,6 +150,7 @@ def main() -> None: processing_class=tokenizer, ) trainer.train() + report_peak_gpu_memory("LoRA training") # measured: ~9 GiB on an A30 at r=16 # Save only the adapter weights (small, ~tens of MB) and the tokenizer. # The base model is NOT duplicated -- load it separately at inference time. diff --git a/code/chapter05/train_qlora.py b/code/chapter05/train_qlora.py index 7ef91b7..b32397c 100644 --- a/code/chapter05/train_qlora.py +++ b/code/chapter05/train_qlora.py @@ -13,7 +13,7 @@ Usage: python -m chapter05.train_qlora \\ --train data/it_support_fmt/train.jsonl \\ - --valid data/it_support/valid.jsonl \\ + --valid data/it_support_fmt/valid.jsonl \\ --out chapter05/runs/it_qlora See Chapter 5, Section 5.7 (QLoRA) and Section 5.1 (Step 5) for details. @@ -33,6 +33,7 @@ from chapter05.dataset import prepare_dataset_for_sft from chapter05.modeling import create_lora_config, load_base_model_qlora, load_tokenizer from common.env import resolve_report_to +from common.gpu import report_peak_gpu_memory from common.seed import seed_everything @@ -156,6 +157,7 @@ def main() -> None: processing_class=tokenizer, ) trainer.train() + report_peak_gpu_memory("QLoRA training") # measured: ~5 GiB on an A30 at r=8 # Save adapter weights and tokenizer. The base model is NOT saved here -- # at inference time, load the base separately with --quantized_4bit and diff --git a/code/common/env.py b/code/common/env.py index 04ead92..9484a90 100644 --- a/code/common/env.py +++ b/code/common/env.py @@ -10,6 +10,29 @@ load_dotenv(_env_path, override=False) +def _guard_hf_transfer() -> None: + """Keep Hugging Face downloads working when HF_HUB_ENABLE_HF_TRANSFER is set + (some cloud images and notebooks export it) but the `hf_transfer` package is + not installed: huggingface_hub then refuses to download at all. The book's + code never needs hf_transfer, so fall back to the standard downloader.""" + flag = os.environ.get("HF_HUB_ENABLE_HF_TRANSFER", "").strip().lower() + if flag not in {"1", "true", "yes", "on"}: + return + try: + import hf_transfer # noqa: F401 + except ImportError: + os.environ["HF_HUB_ENABLE_HF_TRANSFER"] = "0" + import sys + consts = sys.modules.get("huggingface_hub.constants") + if consts is not None: + consts.HF_HUB_ENABLE_HF_TRANSFER = False + print("[common.env] HF_HUB_ENABLE_HF_TRANSFER was set but hf_transfer is not installed; " + "using the standard downloader (pip install hf_transfer to re-enable).") + + +_guard_hf_transfer() + + def env_bool(name: str, default: bool = False) -> bool: v = os.getenv(name) if v is None: diff --git a/code/common/gpu.py b/code/common/gpu.py new file mode 100644 index 0000000..174c49c --- /dev/null +++ b/code/common/gpu.py @@ -0,0 +1,55 @@ +"""Peak GPU-memory reporting shared by every training script. + +Every training entrypoint calls ``report_peak_gpu_memory()`` right after +``trainer.train()`` so each run prints the number that decides whether the +chapter fits your card. The measured values for the book's reference GPU +(NVIDIA A30, 24 GB) live in ``ACCELERATORS.md`` ("GPU requirements at a +glance"); re-measure any chapter with ``python -m scripts.measure_peak_vram``. +""" +from __future__ import annotations + +from typing import Dict, List + +import torch + +GiB = float(2**30) + + +def reset_peak_gpu_memory() -> None: + """Reset the CUDA peak counters (call before training if the model was already loaded).""" + if torch.cuda.is_available(): + for i in range(torch.cuda.device_count()): + torch.cuda.reset_peak_memory_stats(i) + + +def peak_gpu_memory() -> List[Dict[str, object]]: + """Per-device peak allocated / reserved memory in GiB (empty list without CUDA).""" + if not torch.cuda.is_available(): + return [] + return [ + { + "device": i, + "name": torch.cuda.get_device_name(i), + "peak_allocated_gib": round(torch.cuda.max_memory_allocated(i) / GiB, 2), + "peak_reserved_gib": round(torch.cuda.max_memory_reserved(i) / GiB, 2), + } + for i in range(torch.cuda.device_count()) + ] + + +def report_peak_gpu_memory(label: str = "training") -> List[Dict[str, object]]: + """Print and return per-device peaks. With ``device_map="auto"`` on a multi-GPU + box the model is sharded, so the sum across devices is the number to compare + against a single card.""" + stats = peak_gpu_memory() + if not stats: + return stats + print(f"\nPeak GPU memory during {label}:") + for s in stats: + print(f" cuda:{s['device']} ({s['name']}): {s['peak_allocated_gib']:.2f} GiB allocated, " + f"{s['peak_reserved_gib']:.2f} GiB reserved") + if len(stats) > 1: + total = sum(float(s["peak_allocated_gib"]) for s in stats) + print(f" total across {len(stats)} GPUs: {total:.2f} GiB allocated " + f"(this is what a single card would need)") + return stats diff --git a/code/data/README.md b/code/data/README.md index 60aaaac..19eccd0 100644 --- a/code/data/README.md +++ b/code/data/README.md @@ -6,15 +6,17 @@ pipeline in chapters 5 to 8 (LoRA, full SFT, distillation, DPO). | Folder | What it is | Built by | License | | --- | --- | --- | --- | | `it_support/` | Real Stack Exchange IT Q&A (Super User, Ask Ubuntu, Server Fault) as the domain core, plus a small Databricks Dolly slice for general-capability retention. Holds the train / valid / preference splits, the manifest, and per-example source attribution. | `../scripts/build_it_support_dataset.py` | Stack Exchange CC-BY-SA-4.0; Dolly CC-BY-SA-3.0 | -| `it_support_fmt/` | The **same `it_support` data, reformatted** into the assistant's house answer style. `train.jsonl` here is the file the SFT actually trains on. Not a separate dataset, a processed view of `it_support`. | `../scripts/reformat_it_answers.py` | derived from `it_support` | +| `it_support_fmt/` | The **same `it_support` data, reformatted** into the assistant's house answer style (`**Summary:** ... **Steps:** ...`). `train.jsonl` is the file the SFT actually trains on; `valid.jsonl` is the training-time validation split, processed the same way so eval loss is comparable to training loss. Not a separate dataset, a processed view of `it_support`. | `../scripts/reformat_it_answers.py` | derived from `it_support` | Build both from the `code/` directory with: ```bash python scripts/build_it_support_dataset.py # -> data/it_support/ -python scripts/reformat_it_answers.py # -> data/it_support_fmt/train.jsonl +python scripts/reformat_it_answers.py # -> data/it_support_fmt/{train,valid}.jsonl (needs OPENROUTER_API_KEY; both files are committed) ``` +**Two roles for the validation data.** Training scripts take `--valid data/it_support_fmt/valid.jsonl` so the eval loss they report is measured on the same answer style as the training targets. Evaluation scripts (chapter 5 token-F1 and three-lens eval, chapter 8 three-way comparison, chapter 9 drift) take the raw `data/it_support/valid.jsonl` as the held-out test set, so quality is still judged against the original human answers. The prompts are identical in both files; only the assistant text differs. + ## Not the same as `code/contoso_qa_demo/` `code/contoso_qa_demo/` (one level up, not here in `data/`) is a separate, small, hand-authored Contoso Q&A set used only for the chapter 5 section 5.1.8 adapter-beats-prompting demo and the Q&A data-quality example. It is a self-contained mini-project with its own builder, demo, results, and CC0 license. This folder (`data/it_support*`) is the real Stack Exchange corpus the chapters fine-tune on. See [`../contoso_qa_demo/README.md`](../contoso_qa_demo/README.md). diff --git a/code/data/it_support_fmt/valid.jsonl b/code/data/it_support_fmt/valid.jsonl new file mode 100644 index 0000000..8827b1b --- /dev/null +++ b/code/data/it_support_fmt/valid.jsonl @@ -0,0 +1,50 @@ +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "Since Linux has a lot of useful tools, while Windows has a lot of apps (like Chrome), instead of buying another machine to run Linux, is there a way to run it as a Virtual Machine on the PC? The Ubuntu installation CD-ROM doesn't seem to have such an option."}, {"role": "assistant", "content": "**Summary:** You can run Linux on Windows using virtualization software like VMWare, VirtualBox, or VirtualPC, or use lighter-weight options like GNU/Linux tool packages or Cooperative Linux.\n\n**Steps:**\n1. For GNU/Linux tools only, install `cygwin` (bash shell with tools and X11 server), `msys` (lightweight alternative), GNU utilities for Win32 (native versions without DLL dependency), `UWIN` (Unix tools/libraries from ATT Research), or `SUA` (Microsoft's Subsystem for UNIX-based Applications).\n2. For Linux running as a Windows process without full virtualization (32-bit systems only), use Cooperative Linux (`coLinux`) through Portable Ubuntu or `andLinux`.\n3. For full virtualization that lets you boot Linux in a virtual PC with snapshot and suspend/resume capabilities, install `VMWare`, `VirtualBox`, or `VirtualPC`.\n4. For dual-booting (running only one OS at a time), use `wubi` to install Ubuntu from Windows."}], "category": "windows", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "Most Linux software is packaged in tarballs. All of them require but a few commands to compile and install them. \nMy question is; we have gdebi for standalone debian packages, so why not an app to install tarballs the same way? Why is such an apparently simple process not automated? Why must we continue to intimidate and drive away new users with ideas of compiling software?"}, {"role": "assistant", "content": "**Summary:** Tarballs typically contain source code that requires compilation, whereas .deb packages contain pre-compiled binaries, so the compilation step cannot be automated away without creating a package.\n\n**Steps:**\n1. Understand that tarballs generally contain source code and make files, not pre-compiled binaries (though they sometimes contain those too).\n2. Recognize that .deb packages are pre-compiled across multiple architectures, so installation does not require compilation.\n3. Use .deb or .rpm packages when available, as these automate the compilation and installation process.\n4. Use tarballs when you need to access source code, wrap up files for transfer, or when pre-compiled packages are not available for your architecture."}], "category": "linux", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "Where do I go to disable the password complexity policy for the domain?\nI've logged onto the domain controller (Windows Server 2008) and found the option in local policies which is of course locked from any changes. However I can't find the same sort of policies in the group policy manager. Which nodes do I have to expand out to find it?"}, {"role": "assistant", "content": "**Summary:** Edit the \"Default Domain Policy\" in Group Policy Management Console to change the password complexity setting, then run `gpupdate` to apply the change.\n\n**Steps:**\n1. Open Group Policy Management Console by running `GPMC.MSC` from Start / Run.\n2. Expand the Domain node.\n3. Right-click \"Default Domain Policy\" and select Edit.\n4. Navigate to Computer Configuration > Windows Settings > Security Settings > Account Policies.\n5. Modify the password complexity requirements setting.\n6. Run `gpupdate` to apply the change."}], "category": "networking", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "This might be a dumb question but what exactly is the filter rule format to globally allow sites? Do I need multiple entries to handle folder structure of the sites and entries to handle subdomains? Is an asterisk used for a wildcard, do I even need to use a wildcard?"}, {"role": "assistant", "content": "**Summary:** To globally allow a site, add the base URL in the format `example.com Allow`, with no wildcards or directory paths needed.\n\n**Steps:**\n1. Add the base URL to allow an entire site: `example.com Allow`\n2. Add a subdomain URL to allow only that subdomain: `webmail.example.com Allow`\n3. Note that you cannot specify a certain directory or page to allow, as pop-up rules are all based on the site level."}], "category": "networking", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "Inspired by the recent issue with the Gawker network sites, I've been re-evaluating my password management. What strategies do you use to keep passwords safe? I'm particularly interested in software-based solutions available for Ubuntu, but any interesting ideas should be shared. Bonus points for solutions which are available on Android as well."}, {"role": "assistant", "content": "**Summary:** Password Card (http://www.passwordcard.org/en) is a non-software password management solution that provides security.\n\n**Steps:**\n1. Visit http://www.passwordcard.org/en to access Password Card.\n2. Use the card-based system for password management instead of software-based solutions."}], "category": "networking", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I am currently preparing to deploy a JBoss AS into production for the first time. My Linux distribution is Ubuntu Server 8.10 . My original plan was to use aptitude to install JBoss. Currently JBoss AS 4 is the highest version available and my requirements are to use JBoss AS 5. So, I will need to install this manually.\nI ask the following because I am under the belief that most of the automated installs using tools like apt-get and aptitude do all sorts of modifications, different than what you see from downloading the JBoss AS standalone.\n\nWhat is the best way to layout my directory structure?\nWhat about logs, should they map to any particular location?\nWhat is the best way to manage start and stop scripts on Ubuntu Server 8.10 for a Java Application Server?\nAny particular security concerns?"}, {"role": "assistant", "content": "**Summary:** For manual JBoss AS 5 installation on Ubuntu Server 8.10, use `/var/log/` for logs, create a sysvinit script in the appropriate runlevel, and consider that Ubuntu 8.10 only has 12 months of support remaining.\n\n**Steps:**\n1. Place logs in `/var/log/` or a subdirectory within it.\n2. Create an initscript for your runlevel using `run-parts` since Ubuntu 8.10 uses sysvinit.\n3. Consider using a newer Ubuntu version as 8.10 only receives security patches and bug fixes for another 12 months.\n4. To review what an Ubuntu package does differently than manual installation, run `apt-get source package-name`.\n5. Open `package-name.*.diff.gz` and ensure all changes apply to files in `debian/`.\n6. Review the build options in `package-name/debian/rules`.\n7. Review patches to the source itself in `package-name/debian/patches`."}], "category": "linux", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I know that Wubi is slower in disk IO, but I can't find anything about it online other than vague statements such as \"slightly slower disk seeks\". Has anyone done any benchmarks to quantify the speed difference?"}, {"role": "assistant", "content": "**Summary:** Phoronix published benchmarks showing Wubi's performance impact in an article titled \"The Performance Impact Of Ubuntu's Wubi Windows Installer.\"\n\n**Steps:**\n1. Review the benchmark results on pages 2, 3, and 4 of the Phoronix article \"The Performance Impact Of Ubuntu's Wubi Windows Installer\" which quantify the performance differences between Wubi and native Ubuntu installations."}], "category": "hardware", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I have a server with a RAID 1 config. The data on the RAID is in a critical state, how is that different that a rebuild state?\nThe OS is booting up fine and I can sort of read the machines data without error, it is when I start to copy the data that the copy fails.\nAre there any good software tools on the web for RAID analysis?\nThe error I get is from a windows prompt:\n\"Windows - Delay Write Failed\"\n\"Windows was unable to save all the data on the file .... The data has been lost. This error may be caused by a failure of your computer hardware or network connection. Please try to save this file elsewhere.\"\nAny other suggestions as well? \n(The RAID drives are about 2 years old so from what I hear, they may totally fail at the same time.)"}, {"role": "assistant", "content": "**Summary:** A \"critical\" RAID element reports a physical, unrecoverable malfunction and must be replaced as soon as possible, whereas a rebuild state indicates the array is actively reconstructing data onto a replacement drive.\n\n**Steps:**\n1. Since you're running RAID 1, you should be able to run the array degraded and swap your drive out for a new one, if your server must be online all the time.\n2. Consider the possibility that it's your RAID controller card (if you have one) that is malfunctioning, as this is something to keep in mind when rebuilding your array."}], "category": "networking", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "Before I upgraded to Maverick, .swf files used to have a thumbnail in Nautilus.\nHowever, Nautilus doesn't generate thumbnails for them anymore, even after removing `~/.thumbnails/`, which most likely indicates that a package is missing.\nHow do I get Nautilus to generate thumbnails for .swf files?\nEDIT: I've added a bounty, as I believe this question might still be answered..."}, {"role": "assistant", "content": "**Summary:** Install dependencies, compile dump-gnash following the Floorplanner Tech Blog steps with a modified script, then register the thumbnailer with gconf-editor.\n\n**Steps:**\n1. Install required dependencies: `sudo apt-get install gcc libboost-dev libboost-thread-dev libagg-dev libsdl1.2-dev libcairo-dev libgstreamer0.10-dev libatk1.0-dev libglib2.0-dev libpango1.0-dev libgtk2.0-dev libgtkglext1-dev libgl1-mesa-dev libgif-dev libjpeg-dev libgstreamer-plugins-base0.10-dev libspeex-dev libcurl-dev`\n2. Follow the steps in the Floorplanner Tech Blog to compile dump-gnash (this will take a long time).\n3. For Step 7 of the blog instructions, use this script instead:\n```\nif [[ $3 ]]; then\n raw=\"$(mktemp)\"\n dump-gnash $2 -P \"FlashVars=url=file://$1\" -D \"$raw\" --max-advances 1 -j 500 -k 500\n tail -c 1MB \"$raw\" | convert -size 500x500 -depth 8 rgba:- -separate -swap 0,2 -combine -trim png:\"$3\"\n trap \"rm $raw\" EXIT\nelse\n echo \"Insufficient arguments (expected 3 arguments)\"\n exit 1\nfi\n```\n4. Register the thumbnailer with gconf-editor by running these commands:\n```\ngconftool-2 -s \"/desktop/gnome/thumbnailers/application@swf\" -t string \"/usr/bin/swfthumbnailer.sh %s %i %o\"\ngconftool-2 -s \"/desktop/gnome/thumbnailers/application@swf/enable\" -t boolean 'true'\ngconftool-2 -s \"/desktop/gnome/thumbnailers/application@x-swf\" -t string \"/usr/bin/swfthumbnailer.sh %s %i %o\"\ngconftool-2 -s \"/desktop/gnome/thumbnailers/application@x-swf/enable\" -t boolean 'true'\ngconftool-2 -s \"/desktop/gnome/thumbnailers/application@x-shockwave-flash/command\" -t string \"/usr/bin/swfthumbnailer.sh %s %i %o\"\ngconftool-2 -s \"/desktop/gnome/thumbnailers/application@x-shockwave-flash/enable\" -t boolean 'true'\n```\n5. Restart your system or restart `gnome-settings-daemon`."}], "category": "software", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "When you right click on a taskbar icon such as notepad, it will popup a list called \"Recent\".\nIn my case, it is a large annoying list that I would rather have removed. How do I do this?"}, {"role": "assistant", "content": "**Summary:** Modify your Windows 7 jumplist properties to remove the Recent list from taskbar icons.\n\n**Steps:**\n1. Right-click on the taskbar and select \"Properties\"\n2. Navigate to the \"Start Menu\" tab\n3. Uncheck \"Store and display recently opened items in the Start menu and the taskbar\"\n4. Click \"Apply\" and \"OK\""}], "category": "general", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I am creating an authoritative bind server. I have not worked on DNS in a while.\nMy reverse lookups are failing on the internet, but when I connect directly to the server on nslookup I am able to do a reverse lookup sucessfully.\nForward lookups work fine.\nAny ideas on what this could be?"}, {"role": "assistant", "content": "**Summary:** You need to submit a request to ARIN to update the delegated nameservers for your reverse DNS zone so that queries from the internet are directed to your authoritative server.\n\n**Steps:**\n1. Verify that direct queries against your nameserver are working correctly (you've already confirmed this).\n2. Submit a request to ARIN to change the nameserver delegation information for your IP block assignment.\n3. Ensure the nameservers listed in ARIN's records point to your authoritative BIND server for the reverse DNS zone (`xxx.in-addr.arpa`).\n4. Wait for the delegation update to propagate so that queries to the root servers will be directed through the RIR to your nameservers for PTR records."}], "category": "networking", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "Just wondering if there a way to temporarily disable PulseAudio as I am trying to run a specific Wine game (Left 4 Dead 2) and it makes the game crash occasionally.\nIf I kill the process it automatically comes back up. Any suggestions?"}, {"role": "assistant", "content": "**Summary:** Use `pasuspender` to temporarily disable PulseAudio while running the Wine game.\n\n**Steps:**\n1. Prefix your wine command with `pasuspender -- wine path/to/file.exe`"}], "category": "software", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "If you run Task Manager (or Process Explorer) and minimise it, it puts an icon in the icon tray which shows current processor usage as a little bar graph. \nI would like to find a similar tool which will show me current disk activity in a similar way.\nEven better... a tool which shows me a graph of recent disk activity.\nEven better than that ... showing it by application. \nI'm mainly going to use this on Windows XP, but a tool which works for Vista would be useful too. \nI know that some of the Sysinternals tools almost show me this, but none of them is quite there. \nMany thanks for any suggestions."}, {"role": "assistant", "content": "**Summary:** Process Explorer can display I/O usage in the system tray with a tray icon that shows which application is using the most I/O.\n\n**Steps:**\n1. Open Process Explorer\n2. Enable `Options->I/O History Tray Icon`\n3. Hover over the tray icon to see which application is currently using the most I/O"}], "category": "windows", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I moved into a new place and my old Wi-Fi router just doesn't have the range to get into all the rooms. I've been investigating a lot of options and I'm wondering what other folks have done here. Moving the current location of my primary Wi-Fi router is not an option thanks to our cable provider and our landlord.\n\nBuy a bigger, beefier router (seems expensive). If so, should I go for one of those draft 802.11n ones to avoid microwave/other Wi-Fi router interference?\nSet up a router with DD-WRT as a repeater\nLeech the neighbors' open Wi-Fi access point.\n\nAlright, I was kidding about the last one but I'm genuinely curious as to what my best option is."}, {"role": "assistant", "content": "**Summary:** Try making a parabolic reflector from paper and foil to extend your router's range before purchasing new equipment.\n\n**Steps:**\n1. Make a parabolic reflector using paper and foil.\n2. Aim it at the areas that are having difficulty receiving signal.\n3. Note that the reflector will compromise signal slightly on the opposite side of the dish, so position it toward the more important coverage area.\n4. Be aware that using DD-WRT as a repeater with only 1 antenna will cut your speed in half, as the antenna must handle both sending and receiving."}], "category": "networking", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "i want to know how can i possibly keep me secured from forgery sites. i am currently using mozilla 3.7"}, {"role": "assistant", "content": "**Summary:** Use the WOT (Web of Trust) plugin for Firefox to help protect against forgery sites.\n\n**Steps:**\n1. Install the WOT (Web of Trust) plugin for Firefox, which will provide some protection from forgery sites."}], "category": "security", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I have 4GB SD card that I want to put tons of small photos on for a digital photo frame but I cannot copy all the data into the root of the SD card. About half way though the copy it just fails and stops copying.\nI remedied this by creating a sub directory and putting everything in that directory and was able to copy all files just fine but it's just not possible to do so in the root. It got me thinking, is this why digital cameras always save their files to sub directories on the memory cards and not the root directory?\nAny reason why the root would not allow all the files but a sub directory would?"}, {"role": "assistant", "content": "**Summary:** The root directory of FAT file systems (used by SD cards) has a file limit, which is why the copy fails, but subdirectories do not have this limitation.\n\n**Steps:**\n1. Understand that SD cards typically use FAT file systems, which impose a specific file limit on the root directory as documented in http://support.microsoft.com/kb/120138\n2. Store photos in subdirectories instead of the root directory to avoid this limitation\n3. Note that camera manufacturers use subdirectories for this same reason"}], "category": "hardware", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "My games and multimedia PC has twice now been infected with a virus that loads an executable called Ahnrpta.exe. This is with me being the only user and literally not downloading anything and not receiving any email on it.\nI know my housemate has this virus on his laptop (he keeps telling me he'll get some antivirus) so I suspect it's coming from there (we use the same home network). The strange thing is though my work laptop which I use on the same network quite frequently has not been infected.\nBoth my work laptop and my games PC run WinXP and have no firewall other than the WinXP SP2 firewall.\nDoes anyone know how this virus spreads? I haven't been able to find a definitive answer from the antivirus vendor sites."}, {"role": "assistant", "content": "**Summary:** The Ahnrpta.exe virus likely spreads via USB thumbdrives using `autorun.inf` to execute automatically when inserted.\n\n**Steps:**\n1. Disable autorun for external media (CD-ROM, USB drives) by following the referenced tutorial to prevent automatic execution from removable devices."}], "category": "networking", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "What is this process? It on a fresh install of Windows Server 2008 with an Oracle 11g database installed on it. The process is in c:\\windows\\ and according to Sophos Anti-virus it is trying to access the registry. It is probably harmless, but before I take it out of quarantine I'd like to know what it is. Google results only show it appearing in lists of processes."}, {"role": "assistant", "content": "**Summary:** `TIRHService.exe` is the Intuit Track-It! Remoting Helper service from the Track-It helpdesk program.\n\n**Steps:**\n1. The process `TIRHService.exe` stands for Intuit Track-It! Remoting Helper (TIRemotingHelper).\n2. It is part of Intuit Track-It, a helpdesk program used by businesses to create tickets, remote into computers, and perform similar support functions."}], "category": "security", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "How can we set Windows clients to authenticate against an LDAP Server running on Ubuntu?"}, {"role": "assistant", "content": "**Summary:** Use pGina, an open source authentication system that replaces Windows built-in authentication and supports LDAP through plugins.\n\n**Steps:**\n1. Determine what line of pGina to use\n2. Decide what method of authentication you are going to be using (ex: LDAP, RADIUS, FTP, SSH, etc) and download the corresponding plugin\n3. Download pGina from http://www.pgina.org/\n4. Install pGina and the plugin\n5. Configure pGina and the plugin"}], "category": "windows", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "Possible Duplicate:\nWhen reinstalling Windows 7, does the language, version, architecture (64-bit or 32-bit) or source (OEM, retail, or MSDN) matter? \n\nI'm testing Windows 7 professional on VMware, and I came across a setting saying I have 3 days to activate it.\nSuppose I have it activated(on the virtual machine). I'm gonna stop running it virtually at one point and install it on an actual PC/laptop. I'll have to activate again. I wonder if there's any possible trouble I could run into."}, {"role": "assistant", "content": "**Summary:** You can reactivate Windows 7 on a different machine without problems; at worst, you may need to call Microsoft's automated activation line if online activation fails.\n\n**Steps:**\n1. Install Windows 7 on your physical PC/laptop (the same copy previously used in VMware).\n2. Attempt to activate online first.\n3. If online activation fails, call Microsoft's automated help line to activate by phone.\n4. If the automated system doesn't work, a live representative will ask how many machines have this copy installed and provide an activation number as long as it's only installed on one machine."}], "category": "windows", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I was wondering is there a way to play movies in the desktop,like a background?or is it not possible?"}, {"role": "assistant", "content": "**Summary:** You can play movies as desktop background using the method described in the OMG! Ubuntu tutorial.\n\n**Steps:**\n1. Follow the tutorial at `http://www.omgubuntu.co.uk/2010/08/dreamscene-for-ubuntu-ta-very-much/`"}], "category": "general", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "Is there a way to avoid entering a password in the process of booting up?\nI am using a new and latest installation of Ubuntu, with Gnome desktop in it."}, {"role": "assistant", "content": "**Summary:** Enable automatic login through the Login Screen settings in System → Administration.\n\n**Steps:**\n1. Navigate to System → Administration → Login Screen\n2. Unlock the dialog\n3. Choose the account to log in automatically as"}], "category": "security", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I have got a problem with Update Manager. I'm running Lucid Lynx ver. 10.04.2 and I'm unable to upgrade it to 10.10 version.\nI have got this message when trying to upgr. :\n\nThis can be caused by:\n\nUpgrading to a pre-release version of Ubuntu\nRunning the current pre-release version of Ubuntu\nUnofficial software packages not provided by Ubuntu\n\nI couldn't accidentally download pre-released updates or unsupported updates cause both of those options stays 'unticked' in software sources/updates, so that can't be that.\nEDIT: Those options stayed disable. I have never enabled them.\nUnofficial software packages then? If yes, how to find which of them I have to get rid off?\nMy current Ubuntu version is: 10.04.2 LTS\nThanks"}, {"role": "assistant", "content": "**Summary:** Check the log files in `/var/log/dist-upgrade/` to identify what is causing the upgrade failure.\n\n**Steps:**\n1. Review the files in `/var/log/dist-upgrade/` to see where the upgrade is failing.\n2. If you cannot determine the cause from the logs, post the log files for further analysis."}], "category": "linux", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "When I'm at home, is it better to use the laptop plugged into AC power, or with just the battery, for the overall battery life?"}, {"role": "assistant", "content": "**Summary:** Using AC power versus battery makes little difference, but heat shortens battery life; if possible, remove the battery when at home and store it in a cool place.\n\n**Steps:**\n1. Remove the battery from your laptop while at home if possible and store it somewhere cool, as temperature (heat) is the main factor that shortens battery life.\n2. If you have a Li-ion battery, avoid frequent complete discharges but perform a full discharge and recharge approximately every 30th recharge to recalibrate the electronic charge monitor (battery meter).\n3. Charge Li-ion batteries regularly, as they should not be completely discharged frequently.\n4. Refer to these resources for additional battery care tips: http://batteryuniversity.com/learn/article/do_and_dont_battery_table, http://web.archive.org/web/20090811004517/http://www.friedbeef.com/top-15-ways-to-extend-your-laptop-battery-life/, http://en.wikipedia.org/wiki/Lithium-ion_battery#Guidelines_for_prolonging_Li-ion_battery_life, and http://ask-leo.com/how_do_i_maximize_my_battery_life.html."}], "category": "linux", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I just read about this command line Pandora player called pianobar\nIt works great but the homepage says \"last.fm scrobbling (using an external scrobbler)\"\nAny one ever set this up before?"}, {"role": "assistant", "content": "**Summary:** Set up last.fm scrobbling for pianobar by installing `scrobbler-helper`, configuring pianobar's event command, and providing your Last.fm credentials.\n\n**Steps:**\n1. Install the external scrobbler `scrobbler-helper` via CPAN: `sudo perl -MCPAN -e 'install Audio::Scrobbler'`\n2. Create the configuration directory for pianobar (skip if `~/.config/pianobar/config` already exists): `mkdir -p ~/.config/pianobar`\n3. Copy the example config file (skip if `~/.config/pianobar/config` already exists): `cp /usr/share/doc/pianobar/contrib/config-example ~/.config/pianobar/config`\n4. In `~/.config/pianobar/config`, remove the `#` from the beginning of the line `#event_command = /home/user/.config/pianobar/eventcmd` and replace `user` with your actual username\n5. Download the eventcmd script to your Desktop, rename it to `eventcmd`, then run: `cp ~/Desktop/eventcmd ~/.config/pianobar/eventcmd` and `chmod +x ~/.config/pianobar/eventcmd`\n6. Create a file `~/.scrobbler-helper.conf` and paste the following into it, replacing `USERNAME` and `PASSWORD` with your actual Last.fm username and password:\n```\n[global]\nusername=USERNAME\npassword=PASSWORD\n```"}], "category": "general", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I would like to install some Firefox add-ons, also called 'extensions' or 'plugins', like for example Adblock Plus, Ubufox or Greasemonkey.\nShould I install them from the Mozilla website, using the Firefox browser, or should I search Firefox in the Software Center and then select the interesting applications ?\nI suppose the issue is about how the extensions are kept up-to-date. If your answer is 'via the Software Center', how can I migrate well from my former add-ons without losing all my preferences?"}, {"role": "assistant", "content": "**Summary:** Install add-ons either through Firefox's built-in add-on manager for frequent updates or from the repositories for stability, depending on your needs.\n\n**Steps:**\n1. For frequently updated add-ons with major version updates, use the built-in add-on manager in Firefox.\n2. For stable, heavily tested add-ons that only receive security and bugfix updates between major versions, install from the repositories via Software Center.\n3. Install operating-system dependent add-ons (e.g. `ubufox`) from the repositories.\n4. Install security-related add-ons from the repositories.\n5. To migrate from former add-ons, remove the items from the Software Center and your preferences should be kept (though the add-on needs to be aware of previous settings, so it may not always work)."}], "category": "software", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I have a Dell desktop computer with a small SATA hard drive. The drive has two partitions: `C:` and `D:`. `D:` is a recovery partition that can be used to return the machine to its factory default condition.\nI'd like to recover that factory default disk image onto a new large hard drive and use it as my new `C:`. How can I do this?"}, {"role": "assistant", "content": "**Summary:** Recover the factory image on the existing drive first, then use a drive copy utility to transfer and expand partitions to the new larger drive.\n\n**Steps:**\n1. Recover the factory default image onto the existing drive using the `D:` recovery partition.\n2. Use a drive copy and partition expansion utility to copy everything from the existing drive to the new drive, expanding the primary partition as needed.\n3. If the recovery doesn't work after the drive copy (due to Dell's boot sector modifications or drive size information), consult the Dell restore utility documentation at http://www.goodells.net/dellrestore/fixes.htm for information and utilities to modify the boot sector as needed."}], "category": "hardware", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I see Mac OS 10.6 has a new Chinese character input via the trackpad. This looks really cool, but I can't find a detailed description of it, so I'm not sure if I can use it.\n\nCan I use it to enter kanji? I know nothing about Chinese, apart from hearing that kanji uses Chinese characters. But Apple says nothing about kanji or Japanese. Are the characters identical across languages? Is there anything I'll be missing?\nCan I use it to enter kana, too?\nCan I use it on my Macbook? It's a 2GHz Core2Duo, so it's a couple years old, and probably doesn't have the newest trackpad.\n\nThanks!"}, {"role": "assistant", "content": "**Summary:** The Chinese character input recognizes Japanese kanji but not kana, and requires a multitouch trackpad.\n\n**Steps:**\n1. The Chinese character input will recognize Japanese kanji since they are essentially the same characters, but it will not recognize kana.\n2. To check if your MacBook has the required multitouch trackpad, go to System Preferences > Trackpad.\n3. If your trackpad preferences show multitouch options, you can use the feature; otherwise, your trackpad does not support it."}], "category": "general", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "How to increase MTU size on Linux 2.6?\nIs such a thing possible? If yes, which files (in `/etc` or elsewhere) must one edit?"}, {"role": "assistant", "content": "**Summary:** Use `ip link set mtu 4000 dev eth0` or `ifconfig eth0 mtu 4000` to increase MTU size, but only do this if you're certain (e.g., direct server connections), as it can cause packet fragmentation and loss across routed networks.\n\n**Steps:**\n1. Verify this is necessary for your use case (such as two servers directly connected by cross-over cable over Gigabit Ethernet), as PMTUD algorithm already deduces MTU size across the path and increasing MTU can cause packet loss or force routers to fragment packets using CPU instead of ASICs.\n2. Use `ip link set mtu 4000 dev eth0` to set MTU to 4000 bytes (or use `ifconfig eth0 mtu 4000` for the older ifconfig tool).\n3. See http://wiki.archlinux.org/index.php/Jumbo_Frames for additional numbers and hints."}], "category": "linux", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I've been looking into setting up my own email server the past couple of weeks and the more I look into it the more I feel that this is an area I do not want to get into. I guess my heart just isn't into email and everything that comes with it. I just want it to work and forget about it.\nI've heard about people outsourcing their email to Google. I did some searches on this and I've figured out how I can send email using gmail with another email address than my gmail address but, I can't seem to figure out how to go about setting up incoming mail. Is this something you will have to do with Google App Engine?"}, {"role": "assistant", "content": "**Summary:** Route incoming mail to Google Applications (now Google Workspace) by modifying your DNS/Zone Records to point to Google's mail servers.\n\n**Steps:**\n1. Sign up for Google Applications (available in free or paid plans, with paid offering more uptime guarantees and storage).\n2. Modify your Zone Record/DNS settings to route mail from your server to Google (Google provides documentation for this configuration).\n3. Remove or disable your local mail server components (such as `Qmail`, `SpamAssassin`, `Horde`) if desired, as Google will handle all email processing."}], "category": "general", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "On their website, Apple makes a huge deal about their new battery technology that they're using in the MacBook Pro. In actual practice, is the battery performance really better than the previous generation?"}, {"role": "assistant", "content": "**Summary:** According to Anandtech's testing articles, the new MacBook Pro battery performance can be evaluated, with tests conducted on both OS X and Windows (Vista & Windows 7).\n\n**Steps:**\n1. Review the Anandtech article on battery performance for the new MacBook Pro\n2. Review the follow-up Anandtech article that tests battery performance with Vista and Windows 7 as the operating systems"}], "category": "hardware", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I want to stop evolution-data-server-2.24 from loading at startup. I tried to uninstall all evolution related items, but I got some nasty warnings about removing the data server so I left it installed.\nThere are no items in my startup list that look like they could be it. I'm running openSUSE 11.1, if it matters, \nThanks"}, {"role": "assistant", "content": "**Summary:** Evolution Data Server is a required component for GNOME and cannot be safely removed or disabled without breaking other programs.\n\n**Steps:**\n1. Understand that Evolution Data Server provides a single database for common, desktop-wide information, such as a user's address book or calendar events.\n2. Note that it is a dependency of the clock applet from the gnome-applets package, 2.10 release.\n3. Be aware that removing Evolution Data Server would either remove many other programs or completely break your install.\n4. Consider that the memory usage is minimal compared to other panel applets like the volume applet."}], "category": "software", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "We have installed Ubuntu desktop edition on our development server. Now that we have it in a data center we would like to strip it down to a server edition.\nIs there an easy way of doing so rather than just going in and uninstalling packages by hand?"}, {"role": "assistant", "content": "**Summary:** There is no simple automated way to convert Ubuntu desktop to server edition; you must either perform a clean install or manually remove desktop packages and install server components.\n\n**Steps:**\n1. Perform a clean install of Ubuntu server edition, or\n2. Manually remove unneeded desktop packages and install the required server components"}], "category": "linux", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "If you have an iPhone 3G and you upgrade to an iPhone 3GS, can you transfer your apps across phones?"}, {"role": "assistant", "content": "**Summary:** You can transfer your apps by authorizing the new phone on your iTunes account.\n\n**Steps:**\n1. Authorize the new iPhone 3GS on your iTunes account."}], "category": "software", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "UPDATE - Definitely a driver issue. Problem goes away when I uninstall nvidia drivers, comes back when I reinstall them.\nI'm having a really strange hardware problem. I have a workstation with two displays, they're both identical V173 ACER monitors. The workstation has two graphics cards, identical GeForce 6200 cards. Both screens are connected through D-sub cables. However one screen is bright white, and the other is kind of a dull yellow.\nI have controlled for the following factors:\n1) I have tried a third screen on both cards and get the same problem, one will always appear yellow (even the one that was previous bright white).\n2) I have reset both displays to factory defaults.\n3) I have reset all display options in all control panels (the NVIDIA control panel and the windows display panel) to defaults.\n4) I have tried different cables.\n5) I have tried making different monitors primary and secondary.\n6) I have tried another graphics card.\nThe only thing I know for sure is that whichever monitor is set to be the primary monitor will be the yellowish monitor. The other one will always be the bright white one.\nThoughts?"}, {"role": "assistant", "content": "**Summary:** The issue is caused by NVIDIA drivers not properly supporting dual PCI graphics cards, as NVIDIA likely only accounts for PCI-Express + PCI or AGP + PCI configurations.\n\n**Steps:**\n1. The driver does not properly handle dual PCI card setups, causing the primary monitor to display with a yellowish tint while the secondary appears bright white.\n2. Consider using a different driver version or switching to a supported dual-card configuration (PCI-Express + PCI or AGP + PCI) to resolve the color discrepancy."}], "category": "windows", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I would like to create a guest account in Windows Vista (Home Premium version) with the following requirements:\n\nrestrict access to only be able to run installed programs, maybe only run a specific program, e.g. Internet Explorer\ndeny access to browse/use file system\n\nI can create a guest account, but I do not know how to restrict as described above. Is that possible, and if it is, how?"}, {"role": "assistant", "content": "**Summary:** Deny execute permissions on Program Files, then allow specific programs, and remove list folder contents permissions from the file system.\n\n**Steps:**\n1. Deny execute permissions to everything under `Program Files`.\n2. Allow execute permissions for the specific programs you want to allow (e.g., Internet Explorer).\n3. Remove the list folder contents permission from file system locations outside the user profile to prevent browsing."}], "category": "windows", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "When I share files to my home group, I find the the other users are prompted to enter my username/password to access my resource. Is there any way to avoid this?"}, {"role": "assistant", "content": "**Summary:** Grant access rights to the Everyone group on the shared folder to avoid authentication prompts.\n\n**Steps:**\n1. Modify the access rights on the folder to include the Everyone group."}], "category": "security", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I have an account on Windows Small Business Server 2003, for which several users have \"send as\" permission (not \"send on behalf\" but \"send as\"). Each of those users has \"send as\" enabled in active directory, as well as mailbox access enabled in the \"exchange advanced\" tab. \nOne of them I set up a couple of months ago and that account is able to view the mailbox and send emails as the other account fine. The ones I set up today can view the mailbox, but cannot send email as the other account. I receive the error: \"You do not have permission to send to this recipient\".\nExample: If Account A has Send As permission on Account B, I am sending from Account A's outlook with Account B's email address. The error comes back to Account A's inbox.\nWhat am I forgetting to do?"}, {"role": "assistant", "content": "**Summary:** Add the user as a delegate in the mailbox's Outlook settings by creating a profile for the target account and configuring delegation through Tools - Options - Delegates.\n\n**Steps:**\n1. Create a new mail profile for Account B\n2. Open Outlook with this new profile\n3. Go to Tools - Options - Delegates\n4. Add Account A permission to send on behalf of\n5. Test whether Account A can now send on behalf of Account B"}], "category": "security", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I bought the 13\" unibody MacBook one month prior to it's upgrade to become the new improved 13\" unibody MacBook Pro (with unopenable battery compartment, extra 3 hours of battery life, more color gammut, SD slot, firewire 800 port, and a new 8Gb memory limit).\nMy MacBook says it is limited to 4GB of DDR3 1066mhz memory, in 2 SO-DIMMs. But that was written back when you couldn't get 2 4GB SO-DIMMs. Now that you can get them, the very similar MacBook Pro is shipping with 4GB standard, and can be upgraded to 8GB. I've asked (Apple reps), and I'm repeatedly told that my model cannot be upgraded to 8. When I ask for the reason they alway say: \"Because that's the published limit at the time your mac was built.\" I find this unconvincing. If they said: \"Because the memory controller in the chipset is limited to 4GB, despite being seemingly identical to the memory controller in the same chipset newest MacBook model,\" then I'd just take their word for it.\nHas anyone tried it, or found any research as to whether two 4GB DDR3 1066mhz SO-DIMMs can be installed in the unibody MacBook without FireWire?"}, {"role": "assistant", "content": "**Summary:** A firmware update released by Apple for the late 2008 MacBook (Model ID MacBook5,1) expands the memory limit from 4GB to 8GB.\n\n**Steps:**\n1. Check the Boot ROM Version in your System Profiler\n2. Verify your Model ID is `MacBook5,1` and confirm your Boot ROM version is `MB51.007D.B03`\n3. If your Boot ROM version does not match, download `MacBook EFI Firmware Update 1.4` for MacBooks (`MacBook5,1`)\n4. Install the firmware updater according to the instructions\n5. After the update, you can install two 4GB DDR3 1066MHz SO-DIMMs for a total of 8GB RAM"}], "category": "hardware", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "After some hangs on our public web application, it was recommended by a friend that I use Profiler in order to view db activity from each user (including sprocs and duration). Is there a more widely used database server profiler that I could use in order to better pinpoint issues like this?"}, {"role": "assistant", "content": "**Summary:** SQL Server Profiler is sufficient for viewing database activity including stored procedures and duration; no other tool is necessary.\n\n**Steps:**\n1. Run a trace against your server using SQL Server Profiler.\n2. Configure the trace options to narrow down the problem.\n3. Review the trace results to identify issues.\n4. Tweak your trace configuration as needed and rerun the trace multiple times until you see the information you need."}], "category": "software", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "Because one of our customers is running out of disk space, I need to extend his Raid5 with one more disk.\nAt the moment the raid5 contains 3x72gb harddisk. The harddisk i want to add is exactly the same one. \nHe's running Win2003std on a HP Pro Liant DL380 G3\nwhat would be the best (fastest) way? and do i need to convert the partition to dynamic volume? \nanybody tried that with the HP tools?\nkind regards \nsam"}, {"role": "assistant", "content": "**Summary:** Add the new disk through the HP RAID controller BIOS during reboot, wait for the array to rebuild, then expand the partition in Windows.\n\n**Steps:**\n1. Take a backup before making any changes to the storage layer.\n2. Reboot the server and press the appropriate key to enter the RAID controller BIOS.\n3. Add the new drive to the container through the RAID controller interface.\n4. Wait for the array to rebuild completely.\n5. Boot into Windows and expand the partition to use the additional space.\n6. Schedule more downtime than you think you need for this process."}], "category": "hardware", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "Let's scope out an average setup:\n\nWeb company\nAbout 50 users\nA few User Groups (Management, Support, Developers, Sys-Admins, Testers, Sales)\nA bunch of permissions (Only management is allowed to access sensitive business data, only administrators are allowed on production systems, etc. pp.)\nWindows Workstations\nLinux servers\nCommunication (Email, In-house IM, etc.)\nA few Web Apps with native AAA (e.g. Mantis, Mediawiki, etc.)\nA few URLs which need to be protected with Basic Auth\nMaybe a few commercial applications from ISVs with LDAP connectors\n\nGoing with OpenLDAP in such an environment may be feasible but it sure isn't fun, especially when you're new to LDAP. \nA standard answer would be Active Directory (given their Kerberos and LDAP backends they are even somewhat standards conform), but are there any different products which fit nicely in such an environment and might even have an advantage over AD or OpenLDAP?"}, {"role": "assistant", "content": "**Summary:** For a small-to-medium business with these requirements, Active Directory is the recommended solution as it meets all needs, is easy to set up, and has extensive support resources available.\n\n**Steps:**\n1. Active Directory provides all required functionality including Kerberos and LDAP backends for the described environment.\n2. AD is straightforward to set up compared to alternatives like OpenLDAP.\n3. Extensive web-based support material is available for troubleshooting operational issues with AD."}], "category": "security", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "What kind of software is there for receiving faxes from a computer? I want to be able to run it on a server so it's always receiving. Something that flexible with how it stores the faxes, and how it sends notifications would be good. I'd also like a good API or some way of adding custom functionality to it for notifications and such.\nIt will most likely be a Windows server but Linux isn't totally out.\nAny thoughts?"}, {"role": "assistant", "content": "**Summary:** HylaFAX server is a good option that runs on *NIX systems and provides fax receiving capabilities.\n\n**Steps:**\n1. Consider using HylaFAX server for your fax receiving needs.\n2. Note that HylaFAX runs on *NIX operating systems (compatible with your Linux option but not Windows)."}], "category": "windows", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "What popular and tried data recovery options are there?\nI know of these (none of which have GUIs afaik):\n\nforemost - Forensics application to recover data\nscalpel - A Frugal, High Performance File Carver\next3grep - Tool to help recover deleted files on ext3 filesystems\ne2undel - Undelete utility for the ext2 file system\n\nDo you know of any that have desktop frontends? It's a tool you don't get to use much, reading manuals and searching for correct flags is not efficient."}, {"role": "assistant", "content": "**Summary:** TestDisk is a reliable data recovery tool that uses console menus instead of a GUI and has comprehensive documentation.\n\n**Steps:**\n1. Use `testdisk` for data recovery with its console menu interface\n2. Refer to the TestDisk wiki for specific use case guidance and instructions"}], "category": "software", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "How do I find a file on my harddisk that starts with `io_file`?\nI tried: \n\n```\ngrep -r io_file*`\n```\n\nand\n\n```\nfind -name io_file*\n```\n\n`find` did not return anything whereas `grep` seems to take ages without any results.\nWhat am I doing wrong?"}, {"role": "assistant", "content": "**Summary:** Use `find / -name 'io_file*'` with the search string in quotes to prevent shell expansion of the asterisk.\n\n**Steps:**\n1. Run `find / -name 'io_file*'` where `/` specifies the search should start from your entire hard drive.\n2. To search only the current directory and subdirectories, use `find . -name 'io_file*'` instead.\n3. Always quote the search string when it contains shell metacharacters like asterisks, otherwise the shell will parse it before `find` sees it."}], "category": "hardware", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I'd like to drag a few more folders to the places sidebar in nautilus, but want to group these new ones separately to the default folders (like Dropbox, Downloads and Documents). Is there a way to add a new separator to the list?"}, {"role": "assistant", "content": "**Summary:** Create a visual separator by adding a bookmark to an unwanted directory and renaming it to dashes.\n\n**Steps:**\n1. Open Nautilus and browse to a directory you don't want to bookmark, for example `/bin`\n2. From the Bookmarks menu click on \"Add Bookmark\"\n3. Right click on the bookmark for `/bin` and choose \"Rename...\"\n4. Rename the bookmark to `----------`\n5. Close Nautilus and open it again to see the separator-like bookmark"}], "category": "general", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "On my ubuntu machine, I have squid3 as a daemon which starts at boot. \nThe problem is that squid3 takes a long time to start and stop (more than 30 seconds) and it has also slowed down my OS startup/shutdown time considerably.\nHow can I solve this issue?"}, {"role": "assistant", "content": "**Summary:** Reduce the `shutdown_lifetime` parameter in squid3 configuration from its default 30 seconds to a shorter value.\n\n**Steps:**\n1. Check the current `shutdown_lifetime` setting with `grep -B 8 \"# shutdown_lifetime\" /etc/squid3/squid.conf`\n2. Edit `/etc/squid3/squid.conf` and uncomment the `shutdown_lifetime` line\n3. Set a shorter time value, such as `shutdown_lifetime 10 seconds`\n4. Save the configuration file and restart squid3"}], "category": "linux", "source": "askubuntu.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I like to use TreeSize and WinDirStat. What else is out there, and why should I use it?"}, {"role": "assistant", "content": "**Summary:** WinDirStat is a reliable choice that the author consistently returns to after trying other alternatives.\n\n**Steps:**\n1. WinDirStat remains the preferred tool despite testing various other disk space analyzers."}], "category": "general", "source": "serverfault.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I want to see all hidden files on my Mac except `.DS_Store`. Is there a way to do that?"}, {"role": "assistant", "content": "**Summary:** You can show all hidden files in Finder, but there is no way to selectively hide `.DS_Store` files while showing other hidden files.\n\n**Steps:**\n1. Run `defaults write com.apple.finder AppleShowAllFiles -bool YES` in Terminal to show hidden files.\n2. Hold down the option key and right-click the Finder icon in the dock to relaunch Finder.\n3. Note that `.DS_Store` files cannot be hidden separately from other hidden files—you must choose to show all hidden files or none."}], "category": "general", "source": "superuser.com"} +{"messages": [{"role": "system", "content": "You are an IT support assistant. Provide clear, step-by-step answers."}, {"role": "user", "content": "I'm trying to install anti-virus on a server running hyper-v with 3 virtual machines. I've tried one which (shall remain nameless) deleted(!) 2 of the VM's (luckily not the virtual hard disks though) when i re-booted the host and am now wary about trying any others. Can anyone recommend any anti-virus that works well with Hyper-V?\nMany thanks."}, {"role": "assistant", "content": "**Summary:** Do not install antivirus on the Hyper-V host; keep the host OS patched and manage it remotely instead.\n\n**Steps:**\n1. Leave the Hyper-V host OS without antivirus software, as AV can seriously reduce VM speed even when excluding VHD working directories and can disrupt VM networking by scanning network traffic.\n2. Keep the host OS patched with current updates.\n3. Manage the host remotely rather than installing antivirus protection.\n4. Read the article from a Microsoft expert on Virtualisation for additional guidance on this approach."}], "category": "security", "source": "superuser.com"} diff --git a/code/pyproject.toml b/code/pyproject.toml index 1200b47..62821e1 100644 --- a/code/pyproject.toml +++ b/code/pyproject.toml @@ -1,66 +1,68 @@ -[build-system] -requires = ["setuptools>=68", "wheel"] -build-backend = "setuptools.build_meta" - -[project] -name = "model-adaptation-book" -version = "0.1.0" -description = "Runnable code for the book Practical Model Adaptation Techniques for Large Language Models." -readme = "README.md" -requires-python = ">=3.12" -dependencies = [ - "accelerate>=0.30.0", - "datasets>=2.19.0", - "evaluate>=0.4.2", - "huggingface_hub>=0.23.0", - # Capped to pair with the transformers<5.0 cap below: peft<0.18 imports - # HybridCache, which transformers 5.x removed. - "peft>=0.12.0,<0.18", - "rich>=13.7.0", - # safetensors 0.6+ dropped builtins.safe_open, which breaks datasets.map - # fingerprinting via dill. Cap until safetensors restores the symbol or - # datasets stops fingerprinting through it. - "safetensors>=0.4.3,<0.6", - "sentence-transformers>=3.0.0", - # Capped below 5.0: transformers 5.x removed HybridCache, which peft<0.18 - # imports at load time, so a fresh install on 5.x breaks `import peft`. - "transformers>=4.47.0,<5.0", - "trl>=0.9.0", - "python-dotenv>=1.0.1", - # HTML -> text cleaning for the Stack Exchange IT-support dataset builder - # (scripts/build_it_support_dataset.py). - "beautifulsoup4>=4.12.0", -] - -[project.optional-dependencies] -# QLoRA / quantization support. Keep optional for macOS/CPU-only readers. -qlora = ["bitsandbytes>=0.43.0"] - -# Chapter 3 — data-quality experiment and synthetic data pipeline. -# The §3.1 experiment uses transformers + PEFT + TRL (core deps) in bf16 by -# default; bitsandbytes enables the optional 4-bit path. The §3.7 synthetic -# pipeline uses the Anthropic SDK; the experiment plots/scores with matplotlib -# + scikit-learn. -chapter03 = [ - "bitsandbytes>=0.43.0", - "anthropic>=0.30.0", - "matplotlib>=3.8", - "scikit-learn>=1.4", -] - -# Optional experiment tracking. -wandb = ["wandb>=0.16.0"] - -# Dev/test tooling (lightweight). -dev = [ - "pytest>=8.0.0", - "ruff>=0.4.0", -] - -[tool.setuptools.packages.find] -where = ["."] -# Packages: common (shared utilities), chapter05, chapter06, ... (code by chapter) - -[tool.ruff] -line-length = 100 -target-version = "py310" +[build-system] +requires = ["setuptools>=68", "wheel"] +build-backend = "setuptools.build_meta" + +[project] +name = "model-adaptation-book" +version = "0.1.0" +description = "Runnable code for the book Practical Model Adaptation Techniques for Large Language Models." +readme = "README.md" +# Validated on 3.12 (CI matrix). Newer Pythons work once PyTorch ships wheels for them; +# a reader on 3.14 hit "No matching distribution found for torch" before those wheels existed. +requires-python = ">=3.12" +dependencies = [ + "accelerate>=0.30.0", + "datasets>=2.19.0", + "evaluate>=0.4.2", + "huggingface_hub>=0.23.0", + # Capped to pair with the transformers<5.0 cap below: peft<0.18 imports + # HybridCache, which transformers 5.x removed. + "peft>=0.12.0,<0.18", + "rich>=13.7.0", + # safetensors 0.6+ dropped builtins.safe_open, which breaks datasets.map + # fingerprinting via dill. Cap until safetensors restores the symbol or + # datasets stops fingerprinting through it. + "safetensors>=0.4.3,<0.6", + "sentence-transformers>=3.0.0", + # Capped below 5.0: transformers 5.x removed HybridCache, which peft<0.18 + # imports at load time, so a fresh install on 5.x breaks `import peft`. + "transformers>=4.47.0,<5.0", + "trl>=0.9.0", + "python-dotenv>=1.0.1", + # HTML -> text cleaning for the Stack Exchange IT-support dataset builder + # (scripts/build_it_support_dataset.py). + "beautifulsoup4>=4.12.0", +] + +[project.optional-dependencies] +# QLoRA / quantization support. Keep optional for macOS/CPU-only readers. +qlora = ["bitsandbytes>=0.43.0"] + +# Chapter 3 — data-quality experiment and synthetic data pipeline. +# The §3.1 experiment uses transformers + PEFT + TRL (core deps) in bf16 by +# default; bitsandbytes enables the optional 4-bit path. The §3.7 synthetic +# pipeline uses the Anthropic SDK; the experiment plots/scores with matplotlib +# + scikit-learn. +chapter03 = [ + "bitsandbytes>=0.43.0", + "anthropic>=0.30.0", + "matplotlib>=3.8", + "scikit-learn>=1.4", +] + +# Optional experiment tracking. +wandb = ["wandb>=0.16.0"] + +# Dev/test tooling (lightweight). +dev = [ + "pytest>=8.0.0", + "ruff>=0.4.0", +] + +[tool.setuptools.packages.find] +where = ["."] +# Packages: common (shared utilities), chapter05, chapter06, ... (code by chapter) + +[tool.ruff] +line-length = 100 +target-version = "py310" diff --git a/code/scripts/reformat_it_answers.py b/code/scripts/reformat_it_answers.py index 7f51ce4..a843452 100644 --- a/code/scripts/reformat_it_answers.py +++ b/code/scripts/reformat_it_answers.py @@ -12,9 +12,18 @@ Dolly retention examples are left untouched (they are the general-capability slice, not IT-support answers). -Run from code/: - python scripts/reformat_it_answers.py --in data/it_support/train.jsonl \ - --out data/it_support_fmt/train.jsonl [--limit N] +Run from code/. With no arguments it processes BOTH splits: + data/it_support/train.jsonl -> data/it_support_fmt/train.jsonl (training targets) + data/it_support/valid.jsonl -> data/it_support_fmt/valid.jsonl (training-time validation) +The raw data/it_support/valid.jsonl stays the held-out TEST set for the evaluation +scripts, so a model's eval loss is scored on the same answer style it trains on +while its token-F1 is still scored against the original human answers. + + python scripts/reformat_it_answers.py # both splits + python scripts/reformat_it_answers.py --in X --out Y # one file + +Needs OPENROUTER_API_KEY (see code/README.md). Both output files are committed to +the repo, so you only need to run this if you rebuild the dataset from source. """ from __future__ import annotations @@ -69,16 +78,12 @@ def reformat_one(question: str, answer: str) -> str: return (r.get("content") or "").strip() -def main(): - ap = argparse.ArgumentParser() - ap.add_argument("--in", dest="inp", required=True) - ap.add_argument("--out", required=True) - ap.add_argument("--limit", type=int, default=0, help="0 = all") - ap.add_argument("--dry", action="store_true", help="print, do not write") - ap.add_argument("--workers", type=int, default=8, help="concurrent API calls") - args = ap.parse_args() +SPLITS = [("data/it_support/train.jsonl", "data/it_support_fmt/train.jsonl"), + ("data/it_support/valid.jsonl", "data/it_support_fmt/valid.jsonl")] + - rows = list(read_jsonl(args.inp)) +def process(inp: str, out: str, limit: int, dry: bool, workers: int) -> None: + rows = list(read_jsonl(inp)) def qa(row): msgs = row["messages"] @@ -86,7 +91,7 @@ def qa(row): a = next(m["content"] for m in msgs if m["role"] == "assistant") return q, a - if args.dry: + if dry: shown = 0 for row in rows: if row.get("source") == "dolly": @@ -96,15 +101,17 @@ def qa(row): print(f"--- RAW: {a[:200]}") print(f"--- FMT: {reformat_one(q, a)[:400]}") shown += 1 - if shown >= (args.limit or 2): + if shown >= (limit or 2): break return # IT rows to reformat (preserve original index for ordered output). it_idx = [i for i, r in enumerate(rows) if r.get("source") != "dolly"] + if limit: + it_idx = it_idx[:limit] new_answers = {} done = 0 - with ThreadPoolExecutor(max_workers=args.workers) as ex: + with ThreadPoolExecutor(max_workers=workers) as ex: futs = {ex.submit(reformat_one, *qa(rows[i])): i for i in it_idx} for fut in as_completed(futs): i = futs[fut] @@ -127,10 +134,27 @@ def qa(row): else: out_rows.append(row) - Path(args.out).parent.mkdir(parents=True, exist_ok=True) - write_jsonl(args.out, out_rows) + Path(out).parent.mkdir(parents=True, exist_ok=True) + write_jsonl(out, out_rows) n_ok = sum(1 for v in new_answers.values() if v) - print(f"Wrote {args.out}: {len(out_rows)} rows ({n_ok}/{len(it_idx)} IT reformatted, rest passthrough)") + print(f"Wrote {out}: {len(out_rows)} rows ({n_ok}/{len(it_idx)} IT reformatted, rest passthrough)") + + +def main(): + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--in", dest="inp", default=None, + help="input split from build_it_support_dataset.py; with no --in/--out both the train and " + "valid splits are processed (the default the READMEs rely on)") + ap.add_argument("--out", default=None, help="output path (required if --in is given)") + ap.add_argument("--limit", type=int, default=0, help="0 = all") + ap.add_argument("--dry", action="store_true", help="print, do not write") + ap.add_argument("--workers", type=int, default=8, help="concurrent API calls") + args = ap.parse_args() + if (args.inp is None) != (args.out is None): + ap.error("--in and --out must be given together (or neither, to process both default splits)") + pairs = [(args.inp, args.out)] if args.inp else SPLITS + for inp, out in pairs: + process(inp, out, args.limit, args.dry, args.workers) if __name__ == "__main__":