unsloth/docker/unsloth_nb_strip_colab.py
Daniel Han 5906a9feb6 docker: close four holes the review found
docker-publish.yml: the llama.cpp tag resolver was the one step in the prepare
job still using `curl | sed` without pipefail. The runner's default `bash -e`
shell takes sed's exit status, so an unreachable github.com left TAG empty and
the step published the mutable `latest`. Both arch legs re-resolve that through
fetch_llama_prebuilt.py and Dockerfile.studio resolves it a third time, so a
release cut mid-run can put different llama.cpp bundles under one manifest.
Capture the redirect first and fail the job when it is missing or does not land
on a release tag, matching the three ref resolvers below it.

unsloth_nb_strip_colab.py: strip_notebook read, parsed and then unconditionally
os.replace'd. The refresh child re-arms finalize after the entrypoint has execed
the container command, so JupyterLab is already serving the tree and a save
landing in that window was destroyed, after which migrate recorded the cleaned
hash and marked the notebook pristine forever. Re-read the hash once the staged
copy is complete and drop it when the file moved, the same rule the refresh
publish in unsloth_sync_notebooks.sh already follows.

unsloth_nb_view.py: ownership for the view teardown accepted any symlink target
under DEST, but every link the tool creates points at DEST/nb. A shortcut the
user made in the landing dir to their own file elsewhere in the checkout was
therefore classified as ours and deleted on the next boot. Key ownership on
DEST/nb instead.

cellNav.ts: the edit-mode boundary test compared the cursor line against
editor.lineCount, both logical, while JupyterLab wraps markdown and raw editors
by default (StaticNotebook.defaultEditorConfig). A one-line markdown header
renders as several visual rows, so every arrow left the cell and the wrapped
rows could not be reached. Ask CodeMirror whether it can still move one visual
line (EditorView.moveVertically, compared by coordsAtPos top) and keep the
logical test as the fallback for a non-CodeMirror editor.

New tests: 12 passed / 8 failed before, 20 passed / 0 failed after.
2026-07-27 16:06:56 +00:00

242 lines
8.4 KiB
Python

#!/usr/bin/env python3
# SPDX-License-Identifier: AGPL-3.0-only
# Copyright 2026-Present the Unsloth team. See /studio/LICENSE.AGPL-3.0
# Remove the Colab-only "how to run" sentence from Unsloth notebooks for Docker.
#
# Each generated notebook's first markdown cell opens with a Colab instruction
# ("To run this, press Runtime > Run all ...") that is wrong inside Docker. Strip
# only that leading sentence and keep the rest (badge row, install link, etc).
# Docker-only, applied at sync time; NOT pushed upstream.
#
# Two modes:
# unsloth_nb_strip_colab.py <a.ipynb> [b.ipynb ...] strip in place (idempotent)
# unsloth_nb_strip_colab.py --state <STATE> --dest <DEST>
# STATE-aware migration: strip + rehash each owned+unedited notebook (one
# whose hash still matches STATE); user-edited ones are left untouched.
#
# Safe with refresh: content_sig classifies the intro cell as boilerplate, so the
# body digest is unchanged. Exit code is always 0.
import argparse
import hashlib
import json
import os
import sys
# Stable identifier for the offending line (all GPU/Cloud variants).
_INTRO_PREFIX = "to run this, press"
# Baked notebooks ship tqdm widget outputs + a metadata.widgets block that
# JupyterLab can't rebuild, so they render as a stuck "Loading widget...". Drop
# them (the cell recreates a fresh widget). Outputs aren't in the refresh
# signature (content_sig hashes type+source), so this is safe.
_WIDGET_VIEW_MIME = "application/vnd.jupyter.widget-view+json"
def _is_intro_line(line):
"""True for the Colab run announcement in either shipped spelling.
Most notebooks open the line with the sentence itself, but two (NeMo-Gym-*)
ship it inside a single-line HTML comment:
<!-- To run this, press "*Runtime*" ... instance! -->
Only a comment that OPENS AND CLOSES on the same line is matched, so
dropping it can never leave a dangling `<!--` that swallows the rest of the
cell."""
stripped = line.strip()
low = stripped.lower()
if low.startswith(_INTRO_PREFIX):
return True
if low.startswith("<!--") and stripped.endswith("-->"):
return stripped[4:-3].strip().lower().startswith(_INTRO_PREFIX)
return False
def _strip_lines(lines):
"""Drop the intro line (and an immediately-following blank). Return new list
or None if there was nothing to strip."""
for i, line in enumerate(lines):
if _is_intro_line(line):
out = lines[:i] + lines[i + 1 :]
if i < len(out) and out[i].strip() == "":
out = out[:i] + out[i + 1 :]
return out
return None
def _strip_cell(cell):
"""Strip the intro line out of ONE markdown cell. Return True if changed."""
src = cell.get("source")
if isinstance(src, str):
lines = src.splitlines(keepends = True)
as_str = True
elif isinstance(src, list):
lines = list(src)
as_str = False
else:
return False
new_lines = _strip_lines(lines)
if new_lines is None:
return False
cell["source"] = "".join(new_lines) if as_str else new_lines
return True
def _strip_intro(nb):
"""Strip the Colab intro sentence from the LEADING markdown block.
Scanning cells[0] alone missed 23 of the 433 shipped notebooks: 21 put the
Colab badge `<a href=...>` in cells[0] and the sentence in cells[1]
(Advanced_Llama3_2_(3B)_GRPO_LoRA, Falcon_H1-Alpaca, gpt-oss-(20B)-GRPO,
...), and 2 (NeMo-Gym-*) wrap it in an HTML comment cells[0]-only matching
never saw. The scan stops at the first non-markdown cell, so it only ever
touches the header block a notebook opens with (at most 5 cells across the
shipped set) and can never reach explanatory prose between code cells.
Return True if any cell changed."""
cells = nb.get("cells")
if not isinstance(cells, list):
return False
changed = False
for cell in cells:
if not isinstance(cell, dict) or cell.get("cell_type") != "markdown":
break # the first code cell ends the header block
if _strip_cell(cell):
changed = True
return changed
def _clean_widgets(nb):
"""Drop baked ipywidget outputs + the orphan widget-state metadata that
otherwise render as "Loading widget...". Return True if changed."""
changed = False
cells = nb.get("cells")
if isinstance(cells, list):
for cell in cells:
if not isinstance(cell, dict):
continue
outs = cell.get("outputs")
if not isinstance(outs, list):
continue
kept = [
o
for o in outs
if not (isinstance(o, dict) and _WIDGET_VIEW_MIME in (o.get("data") or {}))
]
if len(kept) != len(outs):
cell["outputs"] = kept
changed = True
md = nb.get("metadata")
if isinstance(md, dict) and "widgets" in md:
del md["widgets"]
changed = True
return changed
def strip_notebook(path):
"""Return True if the notebook was modified and written back."""
try:
before = _sha256(path)
with open(path, "r", encoding = "utf-8") as f:
nb = json.load(f)
except Exception:
return False
# Apply both transforms; write back if either changed.
changed = _strip_intro(nb)
changed = _clean_widgets(nb) or changed
if not changed:
return False
tmp = path + ".tmp"
try:
with open(tmp, "w", encoding = "utf-8") as f:
json.dump(nb, f, indent = 1, ensure_ascii = False)
f.write("\n")
# The refresh child re-arms this cleanup AFTER the entrypoint has execed
# the container command, so JupyterLab is already serving the tree: a save
# landing between the read above and this replace would be silently
# overwritten, and migrate() would then record the cleaned hash and mark
# the notebook pristine forever. Re-read the live file once the staged
# copy is complete (the same rule the refresh publish in
# unsloth_sync_notebooks.sh follows) and let their edit win.
if _sha256(path) != before:
os.remove(tmp)
return False
os.replace(tmp, path)
except Exception:
try:
os.remove(tmp)
except OSError:
pass
return False
return True
def _sha256(path):
h = hashlib.sha256()
with open(path, "rb") as f:
for chunk in iter(lambda: f.read(65536), b""):
h.update(chunk)
return h.hexdigest()
def migrate(state_path, dest):
"""Strip owned+unedited notebooks listed in STATE and update their hashes."""
try:
with open(state_path, "r", encoding = "utf-8") as f:
lines = f.read().splitlines()
except OSError:
return 0
out = []
changed = 0
for line in lines:
parts = line.split(" ", 1) # "<sha256> <relpath>"
if len(parts) != 2:
out.append(line)
continue
rec, rel = parts
path = os.path.join(dest, rel)
if rel.endswith(".ipynb") and os.path.isfile(path):
try:
if _sha256(path) == rec: # we own it and it is unedited
if strip_notebook(path):
rec = _sha256(path)
changed += 1
except OSError:
pass
out.append("%s %s" % (rec, rel))
if changed:
tmp = state_path + ".tmp"
try:
with open(tmp, "w", encoding = "utf-8") as f:
f.write("\n".join(out) + "\n")
os.replace(tmp, state_path)
except OSError:
pass
print(f"[unsloth-nb] cleaned {changed} notebook(s) (Colab intro + widget outputs)")
return 0
def main(argv):
ap = argparse.ArgumentParser(description = "Strip the Colab-only intro sentence.")
ap.add_argument("--state", help = "sync state file (enables migration mode)")
ap.add_argument("--dest", help = "notebooks dir (with --state)")
ap.add_argument("paths", nargs = "*", help = "notebooks to strip in place")
args = ap.parse_args(argv)
if args.state:
if not args.dest:
ap.error("--state requires --dest")
return migrate(args.state, args.dest)
changed = sum(1 for p in args.paths if strip_notebook(p))
if changed:
print(f"[unsloth-nb] cleaned {changed} notebook(s) (Colab intro + widget outputs)")
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv[1:]))