From c9ec02c0091906744c5f10c0d373d9444e3a8cd0 Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Fri, 2 Oct 2026 16:36:27 -0700 Subject: [PATCH 01/13] Add DPO e2e validation script for Qwen3.5-9B-Base LoRA Runs the tinker-cookbook DPO recipe (forward_backward_custom + compute_logprobs reference) against a Spindle deployment and compares the final step with the cookbook README's Tinker numbers. Co-Authored-By: Claude Opus 5.5 --- scripts/e2e_dpo_qwen3_5_9b_lora.py | 100 +++++++++++++++++++++++++++++ 1 file changed, 100 insertions(+) create mode 100644 scripts/e2e_dpo_qwen3_5_9b_lora.py diff --git a/scripts/e2e_dpo_qwen3_5_9b_lora.py b/scripts/e2e_dpo_qwen3_5_9b_lora.py new file mode 100644 index 0000000..83b5ecd --- /dev/null +++ b/scripts/e2e_dpo_qwen3_5_9b_lora.py @@ -0,0 +1,100 @@ +# /// script +# requires-python = ">=3.11,<3.13" +# dependencies = [ +# "tinker>=0.24,<0.25", +# "tinker-cookbook @ git+https://github.com/thinking-machines-lab/tinker-cookbook.git@c8ed9c764b59161391156f980102d82f05014765", +# ] +# /// +"""Run the tinker-cookbook DPO recipe against Spindle and compare with Tinker. + +DPO needs no Spindle-specific config: the recipe uses `forward_backward_custom` +(a client-side `forward` plus weighted `cross_entropy`) and computes reference +logprobs with `compute_logprobs` on a sampler published from the step-0 weights. +This runs the cookbook README's example command on the `qwen35-9b-lora-16k` +deployment and prints the last step next to the README's Tinker numbers. + + export TINKER_BASE_URL=https://your-modal-server-url + export TINKER_API_KEY=... + uv run scripts/e2e_dpo_qwen3_5_9b_lora.py --steps 50 +""" + +from __future__ import annotations + +import argparse +import json +import math +import os +from pathlib import Path + +from tinker_cookbook.recipes.preference.dpo.train import CLIConfig, cli_main + +# Step 49 of the cookbook DPO README (hhh, Qwen3.5-9B-Base, lr 1e-5, beta 0.1, batch 256). +TINKER_STEP_49 = { + "dpo_loss": 0.690734, + "accuracy": 0.515748, + "margin": 0.005681, + "chosen_reward": 0.008626, + "rejected_reward": 0.002946, + "time/step": 5.270600, + "time/get_ref_logprobs": 2.125185, +} + + +def read_metrics(log_path: Path) -> list[dict]: + for path in sorted(log_path.rglob("*.jsonl")): + rows = [json.loads(line) for line in path.read_text().splitlines() if line] + if any("dpo_loss" in row for row in rows): + return [row for row in rows if "dpo_loss" in row] + raise FileNotFoundError(f"no DPO metrics under {log_path}") + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--base-model", default="Qwen/Qwen3.5-9B-Base") + parser.add_argument("--dataset", default="hhh") + parser.add_argument("--steps", type=int, default=50) + parser.add_argument("--learning-rate", type=float, default=1e-5) + parser.add_argument("--dpo-beta", type=float, default=0.1) + parser.add_argument("--batch-size", type=int, default=256) + parser.add_argument("--log-path", default="/tmp/spindle-dpo") + parser.add_argument("--wandb-project", default=None) + parser.add_argument("--wandb-name", default=None) + args = parser.parse_args() + + cli_main( + CLIConfig( + model_name=args.base_model, + dataset=args.dataset, + renderer_name="role_colon", + learning_rate=args.learning_rate, + dpo_beta=args.dpo_beta, + batch_size=args.batch_size, + max_steps=args.steps, + log_path=args.log_path, + wandb_project=args.wandb_project, + wandb_name=args.wandb_name, + base_url=os.environ["TINKER_BASE_URL"], + behavior_if_log_dir_exists="delete", + ) + ) + + rows = read_metrics(Path(args.log_path)) + first, last = rows[0], rows[-1] + print( + f"\n{'metric':<24}{'step 0':>12}{f'step {len(rows) - 1}':>12}{'Tinker 49':>12}" + ) + for key, reference in TINKER_STEP_49.items(): + print( + f"{key:<24}{first.get(key, math.nan):>12.6f}{last.get(key, math.nan):>12.6f}{reference:>12.6f}" + ) + + # At this learning rate DPO moves slowly; the README run ends just under ln 2. + learned = last["dpo_loss"] < math.log(2) and last["margin"] > 0 + print( + f"\nDPO {'learned' if learned else 'did NOT learn'}: final dpo_loss {last['dpo_loss']:.6f} vs ln2 {math.log(2):.6f}" + ) + raise SystemExit(0 if learned else 1) + + +if __name__ == "__main__": + main() From 7987c13358b84e4691a2d3c68f41bd523cbc42c2 Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Fri, 2 Oct 2026 16:42:23 -0700 Subject: [PATCH 02/13] Fall back to TCP_KEEPALIVE so spindle deploy works from macOS spindle deploy imports spindle.inference.sampling locally, and macOS has no socket.TCP_KEEPIDLE, so the frontend deploy failed with AttributeError. Co-Authored-By: Claude Opus 5.5 --- src/spindle/inference/sampling.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/src/spindle/inference/sampling.py b/src/spindle/inference/sampling.py index 74fc3fa..df0978e 100644 --- a/src/spindle/inference/sampling.py +++ b/src/spindle/inference/sampling.py @@ -19,7 +19,12 @@ RETRY_MAX_DELAY_SECONDS = 5.0 TCP_KEEPALIVE_OPTIONS = ( (socket.SOL_SOCKET, socket.SO_KEEPALIVE, 1), - (socket.IPPROTO_TCP, socket.TCP_KEEPIDLE, 60), + # macOS names TCP_KEEPIDLE TCP_KEEPALIVE; deploys import this module locally. + ( + socket.IPPROTO_TCP, + getattr(socket, "TCP_KEEPIDLE", None) or socket.TCP_KEEPALIVE, + 60, + ), (socket.IPPROTO_TCP, socket.TCP_KEEPINTVL, 60), (socket.IPPROTO_TCP, socket.TCP_KEEPCNT, 5), ) From 47ad512cb577e5936304a6d20a2df8a34c9e6ccb Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Fri, 2 Oct 2026 16:46:20 -0700 Subject: [PATCH 03/13] Generate tml- prefixed Spindle API keys in the README The Tinker SDK rejects API keys without the tml- prefix, so keys generated with the previous command fail at ServiceClient creation. Co-Authored-By: Claude Opus 5.5 --- README.md | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index c7399bf..9df1895 100644 --- a/README.md +++ b/README.md @@ -58,7 +58,8 @@ Spindle uses three separate credentials: Generate an API key for your Spindle server and store it as a [Modal Secret](https://modal.com/docs/guide/secrets): ```bash -export TINKER_API_KEY="$(openssl rand -hex 32)" +# The Tinker SDK only accepts keys that start with "tml-". +export TINKER_API_KEY="tml-$(openssl rand -hex 32)" modal secret create spindle-api TINKER_API_KEY="$TINKER_API_KEY" ``` From edaabb9d8f700e8663a3ebe06a5b3e1423282d5a Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Fri, 2 Oct 2026 16:48:00 -0700 Subject: [PATCH 04/13] Default the DPO validation to 10 steps Co-Authored-By: Claude Opus 5.5 --- scripts/e2e_dpo_qwen3_5_9b_lora.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scripts/e2e_dpo_qwen3_5_9b_lora.py b/scripts/e2e_dpo_qwen3_5_9b_lora.py index 83b5ecd..3672ede 100644 --- a/scripts/e2e_dpo_qwen3_5_9b_lora.py +++ b/scripts/e2e_dpo_qwen3_5_9b_lora.py @@ -15,7 +15,7 @@ export TINKER_BASE_URL=https://your-modal-server-url export TINKER_API_KEY=... - uv run scripts/e2e_dpo_qwen3_5_9b_lora.py --steps 50 + uv run scripts/e2e_dpo_qwen3_5_9b_lora.py """ from __future__ import annotations @@ -52,7 +52,7 @@ def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--base-model", default="Qwen/Qwen3.5-9B-Base") parser.add_argument("--dataset", default="hhh") - parser.add_argument("--steps", type=int, default=50) + parser.add_argument("--steps", type=int, default=10) parser.add_argument("--learning-rate", type=float, default=1e-5) parser.add_argument("--dpo-beta", type=float, default=0.1) parser.add_argument("--batch-size", type=int, default=256) From b3d6e63c93872d655ce0cea5b1c0f56e541f0482 Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Fri, 2 Oct 2026 16:51:20 -0700 Subject: [PATCH 05/13] Install wandb for the DPO validation's optional W&B logging Co-Authored-By: Claude Opus 5.5 --- scripts/e2e_dpo_qwen3_5_9b_lora.py | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/e2e_dpo_qwen3_5_9b_lora.py b/scripts/e2e_dpo_qwen3_5_9b_lora.py index 3672ede..f0345f8 100644 --- a/scripts/e2e_dpo_qwen3_5_9b_lora.py +++ b/scripts/e2e_dpo_qwen3_5_9b_lora.py @@ -2,6 +2,7 @@ # requires-python = ">=3.11,<3.13" # dependencies = [ # "tinker>=0.24,<0.25", +# "wandb", # "tinker-cookbook @ git+https://github.com/thinking-machines-lab/tinker-cookbook.git@c8ed9c764b59161391156f980102d82f05014765", # ] # /// From 48c5e90874a4d1cf5c528ed2e437ae3ce789835b Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Fri, 2 Oct 2026 17:22:37 -0700 Subject: [PATCH 06/13] Document DPO validation on Qwen3.5-9B-Base LoRA Co-Authored-By: Claude Opus 5.5 --- docs/lora_validation.md | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/docs/lora_validation.md b/docs/lora_validation.md index a09f54f..1693b55 100644 --- a/docs/lora_validation.md +++ b/docs/lora_validation.md @@ -94,3 +94,30 @@ samples, lr 1e-4, PPO clip 0.8/1.28, no KL. Both runs use an 8×H200 trainer and The Spindle curve covers the steps completed at the time of writing; the native Miles curve is the full run. + +## DPO on HHH: Qwen3.5-9B-Base (2026-10-02) + +DPO runs on the stock `qwen35-9b-lora-16k` deployment with no extra config. The +tinker-cookbook DPO recipe uses `forward_backward_custom` (a `forward` followed by +`cross_entropy` with per-token weights) and gets reference logprobs from +`compute_logprobs` on a sampler published from the step-0 weights. Both runs use +the cookbook README settings (HHH, rank 32, β 0.1, batch 256 pairs, linear LR +decay) for 10 steps; the 1e-4 run raises the learning rate to make learning +visible in 10 steps. Reproduce with +[e2e_dpo_qwen3_5_9b_lora.py](../scripts/e2e_dpo_qwen3_5_9b_lora.py). + +| Metric | lr 1e-5, step 0 → 9 | lr 1e-4, step 0 → 9 | Tinker lr 1e-5, step 49 | +| --- | --- | --- | --- | +| dpo_loss | 0.6943 → 0.6930 | 0.6939 → 0.6795 | 0.6907 | +| accuracy | 0.468 → 0.518 | 0.480 → 0.557 | 0.516 | +| margin | −0.0020 → 0.0010 | −0.0011 → 0.0502 | 0.0057 | +| warm step time | ~20 s + 13–17 s reference | ~21 s + 9–14 s reference | 5.3 s + 2.1 s reference | + +- At lr 1e-5 the metrics after 10 steps sit in the same range as Tinker's after 50, + but the change is too small to separate from noise. At lr 1e-4 the margin grows + to 0.05 and accuracy to 0.56 by step 9. +- Reference logprobs go through the SGLang sampler pool, which is most of the gap + to Tinker's step time. A built-in `dpo` loss that takes reference logprobs from + the trainer with the adapter disabled would remove that pool round trip. +- W&B: [lr 1e-5](https://wandb.ai/modal-labs/spindle-dpo-validation/runs/jhdhr6mr), + [lr 1e-4](https://wandb.ai/modal-labs/spindle-dpo-validation/runs/0z0obns5). From e990f43759ebb9750b860a90f450846bfaf3094e Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Sun, 4 Oct 2026 14:42:42 -0700 Subject: [PATCH 07/13] Simplify keepalive and API key comments Co-Authored-By: Claude Opus 5.5 --- README.md | 1 - src/spindle/inference/sampling.py | 2 +- 2 files changed, 1 insertion(+), 2 deletions(-) diff --git a/README.md b/README.md index 9df1895..e334d21 100644 --- a/README.md +++ b/README.md @@ -58,7 +58,6 @@ Spindle uses three separate credentials: Generate an API key for your Spindle server and store it as a [Modal Secret](https://modal.com/docs/guide/secrets): ```bash -# The Tinker SDK only accepts keys that start with "tml-". export TINKER_API_KEY="tml-$(openssl rand -hex 32)" modal secret create spindle-api TINKER_API_KEY="$TINKER_API_KEY" ``` diff --git a/src/spindle/inference/sampling.py b/src/spindle/inference/sampling.py index df0978e..88c92aa 100644 --- a/src/spindle/inference/sampling.py +++ b/src/spindle/inference/sampling.py @@ -19,7 +19,7 @@ RETRY_MAX_DELAY_SECONDS = 5.0 TCP_KEEPALIVE_OPTIONS = ( (socket.SOL_SOCKET, socket.SO_KEEPALIVE, 1), - # macOS names TCP_KEEPIDLE TCP_KEEPALIVE; deploys import this module locally. + # TCP configs validated on MacOS and Linux. ( socket.IPPROTO_TCP, getattr(socket, "TCP_KEEPIDLE", None) or socket.TCP_KEEPALIVE, From 4e2c1fafc909c1486bc995b9df9127438eaa497f Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Sun, 4 Oct 2026 14:43:33 -0700 Subject: [PATCH 08/13] Keep only the lr 1e-4 DPO run in the validation docs Co-Authored-By: Claude Opus 5.5 --- docs/lora_validation.md | 42 +++++++++++++++++++++-------------------- 1 file changed, 22 insertions(+), 20 deletions(-) diff --git a/docs/lora_validation.md b/docs/lora_validation.md index 1693b55..f8efee2 100644 --- a/docs/lora_validation.md +++ b/docs/lora_validation.md @@ -100,24 +100,26 @@ Miles curve is the full run. DPO runs on the stock `qwen35-9b-lora-16k` deployment with no extra config. The tinker-cookbook DPO recipe uses `forward_backward_custom` (a `forward` followed by `cross_entropy` with per-token weights) and gets reference logprobs from -`compute_logprobs` on a sampler published from the step-0 weights. Both runs use +`compute_logprobs` on a sampler published from the step-0 weights. The run uses the cookbook README settings (HHH, rank 32, β 0.1, batch 256 pairs, linear LR -decay) for 10 steps; the 1e-4 run raises the learning rate to make learning -visible in 10 steps. Reproduce with -[e2e_dpo_qwen3_5_9b_lora.py](../scripts/e2e_dpo_qwen3_5_9b_lora.py). - -| Metric | lr 1e-5, step 0 → 9 | lr 1e-4, step 0 → 9 | Tinker lr 1e-5, step 49 | -| --- | --- | --- | --- | -| dpo_loss | 0.6943 → 0.6930 | 0.6939 → 0.6795 | 0.6907 | -| accuracy | 0.468 → 0.518 | 0.480 → 0.557 | 0.516 | -| margin | −0.0020 → 0.0010 | −0.0011 → 0.0502 | 0.0057 | -| warm step time | ~20 s + 13–17 s reference | ~21 s + 9–14 s reference | 5.3 s + 2.1 s reference | - -- At lr 1e-5 the metrics after 10 steps sit in the same range as Tinker's after 50, - but the change is too small to separate from noise. At lr 1e-4 the margin grows - to 0.05 and accuracy to 0.56 by step 9. -- Reference logprobs go through the SGLang sampler pool, which is most of the gap - to Tinker's step time. A built-in `dpo` loss that takes reference logprobs from - the trainer with the adapter disabled would remove that pool round trip. -- W&B: [lr 1e-5](https://wandb.ai/modal-labs/spindle-dpo-validation/runs/jhdhr6mr), - [lr 1e-4](https://wandb.ai/modal-labs/spindle-dpo-validation/runs/0z0obns5). +decay) with lr 1e-4 for 10 steps: + +```bash +uv run scripts/e2e_dpo_qwen3_5_9b_lora.py --steps 10 --learning-rate 1e-4 \ + --wandb-project spindle-dpo-validation +``` + +| Metric | Step 0 | Step 9 | +| --- | --- | --- | +| dpo_loss | 0.6939 | 0.6795 | +| accuracy | 0.480 | 0.557 | +| margin | −0.0011 | 0.0502 | +| chosen reward | −0.0001 | 0.0855 | +| rejected reward | 0.0011 | 0.0353 | + +- Margin and accuracy rise over the run, with chosen rewards pulling away from + rejected ones. +- A warm step takes about 21 s plus 9–14 s for reference logprobs. Reference + logprobs go through the SGLang sampler pool; a built-in `dpo` loss that takes + them from the trainer with the adapter disabled would remove that round trip. +- [W&B run](https://wandb.ai/modal-labs/spindle-dpo-validation/runs/0z0obns5). From 1bc684a3e568bf2f149ba1f1d1b3b9233fa3c6d8 Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Sun, 4 Oct 2026 14:48:02 -0700 Subject: [PATCH 09/13] Trim DPO validation note Co-Authored-By: Claude Opus 5.5 --- docs/lora_validation.md | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/docs/lora_validation.md b/docs/lora_validation.md index f8efee2..e825615 100644 --- a/docs/lora_validation.md +++ b/docs/lora_validation.md @@ -119,7 +119,6 @@ uv run scripts/e2e_dpo_qwen3_5_9b_lora.py --steps 10 --learning-rate 1e-4 \ - Margin and accuracy rise over the run, with chosen rewards pulling away from rejected ones. -- A warm step takes about 21 s plus 9–14 s for reference logprobs. Reference - logprobs go through the SGLang sampler pool; a built-in `dpo` loss that takes - them from the trainer with the adapter disabled would remove that round trip. +- A warm step takes about 21 s, plus 9–14 s to compute reference logprobs on + the sampler pool. - [W&B run](https://wandb.ai/modal-labs/spindle-dpo-validation/runs/0z0obns5). From 30c0bfdeae3cfeba653b2d732d87df64ff61452c Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Sun, 4 Oct 2026 14:50:03 -0700 Subject: [PATCH 10/13] Link the cookbook DPO README instead of hard-coding Tinker metrics Co-Authored-By: Claude Opus 5.5 --- scripts/e2e_dpo_qwen3_5_9b_lora.py | 48 ++++-------------------------- 1 file changed, 5 insertions(+), 43 deletions(-) diff --git a/scripts/e2e_dpo_qwen3_5_9b_lora.py b/scripts/e2e_dpo_qwen3_5_9b_lora.py index f0345f8..a19d9ed 100644 --- a/scripts/e2e_dpo_qwen3_5_9b_lora.py +++ b/scripts/e2e_dpo_qwen3_5_9b_lora.py @@ -6,13 +6,14 @@ # "tinker-cookbook @ git+https://github.com/thinking-machines-lab/tinker-cookbook.git@c8ed9c764b59161391156f980102d82f05014765", # ] # /// -"""Run the tinker-cookbook DPO recipe against Spindle and compare with Tinker. +"""Run the tinker-cookbook DPO recipe against Spindle. DPO needs no Spindle-specific config: the recipe uses `forward_backward_custom` (a client-side `forward` plus weighted `cross_entropy`) and computes reference logprobs with `compute_logprobs` on a sampler published from the step-0 weights. -This runs the cookbook README's example command on the `qwen35-9b-lora-16k` -deployment and prints the last step next to the README's Tinker numbers. +This runs the recipe on the `qwen35-9b-lora-16k` deployment. For the recipe and +Tinker's reference metrics, see +https://github.com/thinking-machines-lab/tinker-cookbook/blob/c8ed9c764b59161391156f980102d82f05014765/tinker_cookbook/recipes/preference/dpo/README.md export TINKER_BASE_URL=https://your-modal-server-url export TINKER_API_KEY=... @@ -22,39 +23,17 @@ from __future__ import annotations import argparse -import json -import math import os -from pathlib import Path from tinker_cookbook.recipes.preference.dpo.train import CLIConfig, cli_main -# Step 49 of the cookbook DPO README (hhh, Qwen3.5-9B-Base, lr 1e-5, beta 0.1, batch 256). -TINKER_STEP_49 = { - "dpo_loss": 0.690734, - "accuracy": 0.515748, - "margin": 0.005681, - "chosen_reward": 0.008626, - "rejected_reward": 0.002946, - "time/step": 5.270600, - "time/get_ref_logprobs": 2.125185, -} - - -def read_metrics(log_path: Path) -> list[dict]: - for path in sorted(log_path.rglob("*.jsonl")): - rows = [json.loads(line) for line in path.read_text().splitlines() if line] - if any("dpo_loss" in row for row in rows): - return [row for row in rows if "dpo_loss" in row] - raise FileNotFoundError(f"no DPO metrics under {log_path}") - def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--base-model", default="Qwen/Qwen3.5-9B-Base") parser.add_argument("--dataset", default="hhh") parser.add_argument("--steps", type=int, default=10) - parser.add_argument("--learning-rate", type=float, default=1e-5) + parser.add_argument("--learning-rate", type=float, default=1e-4) parser.add_argument("--dpo-beta", type=float, default=0.1) parser.add_argument("--batch-size", type=int, default=256) parser.add_argument("--log-path", default="/tmp/spindle-dpo") @@ -79,23 +58,6 @@ def main() -> None: ) ) - rows = read_metrics(Path(args.log_path)) - first, last = rows[0], rows[-1] - print( - f"\n{'metric':<24}{'step 0':>12}{f'step {len(rows) - 1}':>12}{'Tinker 49':>12}" - ) - for key, reference in TINKER_STEP_49.items(): - print( - f"{key:<24}{first.get(key, math.nan):>12.6f}{last.get(key, math.nan):>12.6f}{reference:>12.6f}" - ) - - # At this learning rate DPO moves slowly; the README run ends just under ln 2. - learned = last["dpo_loss"] < math.log(2) and last["margin"] > 0 - print( - f"\nDPO {'learned' if learned else 'did NOT learn'}: final dpo_loss {last['dpo_loss']:.6f} vs ln2 {math.log(2):.6f}" - ) - raise SystemExit(0 if learned else 1) - if __name__ == "__main__": main() From adba1cfd2ac7c5d10fa992f22fecec0482f74d14 Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Sun, 4 Oct 2026 14:54:53 -0700 Subject: [PATCH 11/13] Document DPO validation with the stock tinker-cookbook command Spindle works as a drop-in Tinker backend URL, so the validation uses the cookbook's DPO recipe directly instead of a wrapper script. Co-Authored-By: Claude Opus 5.5 --- docs/lora_validation.md | 16 +++++--- scripts/e2e_dpo_qwen3_5_9b_lora.py | 63 ------------------------------ 2 files changed, 11 insertions(+), 68 deletions(-) delete mode 100644 scripts/e2e_dpo_qwen3_5_9b_lora.py diff --git a/docs/lora_validation.md b/docs/lora_validation.md index e825615..065abe1 100644 --- a/docs/lora_validation.md +++ b/docs/lora_validation.md @@ -100,13 +100,19 @@ Miles curve is the full run. DPO runs on the stock `qwen35-9b-lora-16k` deployment with no extra config. The tinker-cookbook DPO recipe uses `forward_backward_custom` (a `forward` followed by `cross_entropy` with per-token weights) and gets reference logprobs from -`compute_logprobs` on a sampler published from the step-0 weights. The run uses -the cookbook README settings (HHH, rank 32, β 0.1, batch 256 pairs, linear LR -decay) with lr 1e-4 for 10 steps: +`compute_logprobs` on a sampler published from the step-0 weights. The run is the +[cookbook DPO recipe](https://github.com/thinking-machines-lab/tinker-cookbook/blob/c8ed9c764b59161391156f980102d82f05014765/tinker_cookbook/recipes/preference/dpo/README.md) +pointed at a Spindle server, with lr 1e-4 for 10 steps: ```bash -uv run scripts/e2e_dpo_qwen3_5_9b_lora.py --steps 10 --learning-rate 1e-4 \ - --wandb-project spindle-dpo-validation +export TINKER_BASE_URL=https://your-modal-server-url +export TINKER_API_KEY=... +uv run --python 3.12 --with "tinker>=0.24,<0.25" \ + --with "tinker-cookbook @ git+https://github.com/thinking-machines-lab/tinker-cookbook.git@c8ed9c764b59161391156f980102d82f05014765" \ + python -m tinker_cookbook.recipes.preference.dpo.train \ + base_url=$TINKER_BASE_URL model_name=Qwen/Qwen3.5-9B-Base dataset=hhh \ + renderer_name=role_colon learning_rate=1e-4 dpo_beta=0.1 max_steps=10 \ + log_path=/tmp/dpo-hhh-experiment ``` | Metric | Step 0 | Step 9 | diff --git a/scripts/e2e_dpo_qwen3_5_9b_lora.py b/scripts/e2e_dpo_qwen3_5_9b_lora.py deleted file mode 100644 index a19d9ed..0000000 --- a/scripts/e2e_dpo_qwen3_5_9b_lora.py +++ /dev/null @@ -1,63 +0,0 @@ -# /// script -# requires-python = ">=3.11,<3.13" -# dependencies = [ -# "tinker>=0.24,<0.25", -# "wandb", -# "tinker-cookbook @ git+https://github.com/thinking-machines-lab/tinker-cookbook.git@c8ed9c764b59161391156f980102d82f05014765", -# ] -# /// -"""Run the tinker-cookbook DPO recipe against Spindle. - -DPO needs no Spindle-specific config: the recipe uses `forward_backward_custom` -(a client-side `forward` plus weighted `cross_entropy`) and computes reference -logprobs with `compute_logprobs` on a sampler published from the step-0 weights. -This runs the recipe on the `qwen35-9b-lora-16k` deployment. For the recipe and -Tinker's reference metrics, see -https://github.com/thinking-machines-lab/tinker-cookbook/blob/c8ed9c764b59161391156f980102d82f05014765/tinker_cookbook/recipes/preference/dpo/README.md - - export TINKER_BASE_URL=https://your-modal-server-url - export TINKER_API_KEY=... - uv run scripts/e2e_dpo_qwen3_5_9b_lora.py -""" - -from __future__ import annotations - -import argparse -import os - -from tinker_cookbook.recipes.preference.dpo.train import CLIConfig, cli_main - - -def main() -> None: - parser = argparse.ArgumentParser() - parser.add_argument("--base-model", default="Qwen/Qwen3.5-9B-Base") - parser.add_argument("--dataset", default="hhh") - parser.add_argument("--steps", type=int, default=10) - parser.add_argument("--learning-rate", type=float, default=1e-4) - parser.add_argument("--dpo-beta", type=float, default=0.1) - parser.add_argument("--batch-size", type=int, default=256) - parser.add_argument("--log-path", default="/tmp/spindle-dpo") - parser.add_argument("--wandb-project", default=None) - parser.add_argument("--wandb-name", default=None) - args = parser.parse_args() - - cli_main( - CLIConfig( - model_name=args.base_model, - dataset=args.dataset, - renderer_name="role_colon", - learning_rate=args.learning_rate, - dpo_beta=args.dpo_beta, - batch_size=args.batch_size, - max_steps=args.steps, - log_path=args.log_path, - wandb_project=args.wandb_project, - wandb_name=args.wandb_name, - base_url=os.environ["TINKER_BASE_URL"], - behavior_if_log_dir_exists="delete", - ) - ) - - -if __name__ == "__main__": - main() From 49e4bb680ee151624b3c2064c64d12459569fa6e Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Sun, 4 Oct 2026 14:57:03 -0700 Subject: [PATCH 12/13] Use the unpinned cookbook command in the DPO validation docs Co-Authored-By: Claude Opus 5.5 --- docs/lora_validation.md | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/docs/lora_validation.md b/docs/lora_validation.md index 065abe1..f1659d6 100644 --- a/docs/lora_validation.md +++ b/docs/lora_validation.md @@ -101,15 +101,13 @@ DPO runs on the stock `qwen35-9b-lora-16k` deployment with no extra config. The tinker-cookbook DPO recipe uses `forward_backward_custom` (a `forward` followed by `cross_entropy` with per-token weights) and gets reference logprobs from `compute_logprobs` on a sampler published from the step-0 weights. The run is the -[cookbook DPO recipe](https://github.com/thinking-machines-lab/tinker-cookbook/blob/c8ed9c764b59161391156f980102d82f05014765/tinker_cookbook/recipes/preference/dpo/README.md) +[cookbook DPO recipe](https://github.com/thinking-machines-lab/tinker-cookbook/blob/main/tinker_cookbook/recipes/preference/dpo/README.md) pointed at a Spindle server, with lr 1e-4 for 10 steps: ```bash export TINKER_BASE_URL=https://your-modal-server-url export TINKER_API_KEY=... -uv run --python 3.12 --with "tinker>=0.24,<0.25" \ - --with "tinker-cookbook @ git+https://github.com/thinking-machines-lab/tinker-cookbook.git@c8ed9c764b59161391156f980102d82f05014765" \ - python -m tinker_cookbook.recipes.preference.dpo.train \ +python -m tinker_cookbook.recipes.preference.dpo.train \ base_url=$TINKER_BASE_URL model_name=Qwen/Qwen3.5-9B-Base dataset=hhh \ renderer_name=role_colon learning_rate=1e-4 dpo_beta=0.1 max_steps=10 \ log_path=/tmp/dpo-hhh-experiment From a66c7337cc719785b1378f148090de0e65f931c3 Mon Sep 17 00:00:00 2001 From: Micah Yong Date: Sun, 4 Oct 2026 14:59:13 -0700 Subject: [PATCH 13/13] Link the HHH dataset in the DPO validation command Co-Authored-By: Claude Opus 5.5 --- docs/lora_validation.md | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/lora_validation.md b/docs/lora_validation.md index f1659d6..12cde6d 100644 --- a/docs/lora_validation.md +++ b/docs/lora_validation.md @@ -107,6 +107,7 @@ pointed at a Spindle server, with lr 1e-4 for 10 steps: ```bash export TINKER_BASE_URL=https://your-modal-server-url export TINKER_API_KEY=... +# dataset=hhh: https://huggingface.co/datasets/Anthropic/hh-rlhf python -m tinker_cookbook.recipes.preference.dpo.train \ base_url=$TINKER_BASE_URL model_name=Qwen/Qwen3.5-9B-Base dataset=hhh \ renderer_name=role_colon learning_rate=1e-4 dpo_beta=0.1 max_steps=10 \