Download abliteration/run_quantize_swift15_375.py from tiktits/Swift-1.5-Qwen3.8-27B-Uncensored-EXL3-3.75bpw: direct link, hf CLI and curl.
- Browser
- Download file 9.94 kB
-
https://huggingface.co/tiktits/Swift-1.5-Qwen3.8-27B-Uncensored-EXL3-3.75bpw/resolve/main/abliteration/run_quantize_swift15_375.py
- Command line
-
hf download hf://tiktits/Swift-1.5-Qwen3.8-27B-Uncensored-EXL3-3.75bpw/abliteration/run_quantize_swift15_375.py
-
curl -L -o run_quantize_swift15_375.py https://huggingface.co/tiktits/Swift-1.5-Qwen3.8-27B-Uncensored-EXL3-3.75bpw/resolve/main/abliteration/run_quantize_swift15_375.py
9.94 kB
| import os | |
| os.environ["PYTORCH_CUDA_ALLOC_CONF"] = "max_split_size_mb:128" | |
| import sys | |
| import json | |
| import time | |
| import shutil | |
| import subprocess | |
| import urllib.request | |
| from pathlib import Path | |
| import torch | |
| from safetensors import safe_open | |
| from safetensors.torch import save_file | |
| from huggingface_hub import snapshot_download | |
| BF16_DIR = r"D:\Documentos\LLama\models\Swift-1.5-Qwen3.8-27B-Uncensored-BF16" | |
| WORK_DIR = r"D:\Documentos\LLama\work_swift15_3.75bpw" | |
| OUT_DIR = r"D:\Documentos\LLama\models\Swift-1.5-Qwen3.8-27B-Uncensored-EXL3-3.75bpw" | |
| RECIPE = r"D:\Documentos\LLama\cal_qwen38\recipe_3.75bpw_H5.yaml" | |
| CAL_DATA = r"D:\Documentos\LLama\cal_qwen38\cal_trace.safetensors" | |
| CONVERT_PY = r"D:\Documentos\LLama\convert.py" | |
| ABLIT_DIR = r"D:\Documentos\LLama\abliteration_swift15\abliteration" | |
| R_PT_PATH = os.path.join(ABLIT_DIR, "r.pt") | |
| RECOVER_JSON_PATH = os.path.join(ABLIT_DIR, "recover.json") | |
| ABLIT_MARKER = os.path.join(BF16_DIR, ".abliterated_done") | |
| def free_comfyui_vram_if_idle(): | |
| """If GPU VRAM is held by an idle ComfyUI instance on port 8188, ask it to unload cached models cleanly.""" | |
| try: | |
| free_b, total_b = torch.cuda.mem_get_info(0) | |
| free_gib = free_b / (1024 ** 3) | |
| print(f" -- Current GPU VRAM free: {free_gib:.2f} GiB / {total_b / (1024**3):.2f} GiB", flush=True) | |
| if free_gib >= 16.0: | |
| return | |
| print(" -- Checking if ComfyUI (127.0.0.1:8188) is idle to release cached VRAM...", flush=True) | |
| while True: | |
| try: | |
| with urllib.request.urlopen("http://127.0.0.1:8188/queue", timeout=5) as resp: | |
| q = json.loads(resp.read().decode("utf-8")) | |
| running = q.get("queue_running", []) | |
| pending = q.get("queue_pending", []) | |
| if not running and not pending: | |
| req = urllib.request.Request( | |
| "http://127.0.0.1:8188/free", | |
| data=json.dumps({"unload_models": True, "free_memory": True}).encode("utf-8"), | |
| headers={"Content-Type": "application/json"}, | |
| method="POST", | |
| ) | |
| urllib.request.urlopen(req, timeout=10) | |
| time.sleep(3) | |
| free_b2, _ = torch.cuda.mem_get_info(0) | |
| print(f" -- ComfyUI unloaded idle models. Free GPU VRAM is now {free_b2 / (1024**3):.2f} GiB!", flush=True) | |
| break | |
| else: | |
| print(" -- ComfyUI is actively generating a job; waiting 15s before checking again...", flush=True) | |
| time.sleep(15) | |
| except Exception: | |
| break | |
| except Exception as e: | |
| print(f" -- Note during VRAM check: {e}", flush=True) | |
| def run_abliteration_in_place(ckpt_dir: str): | |
| """Apply ajgazin/orcarouter rank-1 refusal orthogonalization (131 tensors) in float32 on CPU.""" | |
| if os.path.isfile(ABLIT_MARKER): | |
| print(" -- [Step 2/3] Refusal abliteration (.abliterated_done) already applied! Skipping.", flush=True) | |
| return | |
| print("=========================================================================", flush=True) | |
| print(" [Step 2/3] Applying Single-Direction Refusal Abliteration (131 tensors)", flush=True) | |
| print(" W' = W - r (r^T W) | E' = E - (E r) r^T (float32 -> BF16)", flush=True) | |
| print("=========================================================================", flush=True) | |
| r = torch.load(R_PT_PATH, map_location="cpu")["r"].to(dtype=torch.float32) | |
| r = r / r.norm() | |
| with open(RECOVER_JSON_PATH, "r", encoding="utf-8") as f: | |
| changed_names = [e["name"] for e in json.load(f)["changed"]] | |
| index_file = os.path.join(ckpt_dir, "model.safetensors.index.json") | |
| with open(index_file, "r", encoding="utf-8") as f: | |
| wmap = json.load(f)["weight_map"] | |
| by_shard: dict[str, list[str]] = {} | |
| for name in sorted(changed_names): | |
| if name not in wmap: | |
| raise RuntimeError(f"Expected tensor {name} not found in {index_file}") | |
| by_shard.setdefault(wmap[name], []).append(name) | |
| t0 = time.time() | |
| total_edited = 0 | |
| for idx, (shard, shard_names) in enumerate(sorted(by_shard.items()), 1): | |
| shard_path = os.path.join(ckpt_dir, shard) | |
| shard_done_marker = shard_path + ".ablit_done" | |
| if os.path.isfile(shard_done_marker): | |
| print(f" [{idx}/{len(by_shard)}] {shard}: already abliterated ({len(shard_names)} tensors)", flush=True) | |
| total_edited += len(shard_names) | |
| continue | |
| s_t0 = time.time() | |
| tensors = {} | |
| with safe_open(shard_path, framework="pt", device="cpu") as f: | |
| metadata = f.metadata() | |
| for k in f.keys(): | |
| tensors[k] = f.get_tensor(k) | |
| for name in shard_names: | |
| w = tensors[name] | |
| w32 = w.to(dtype=torch.float32) | |
| is_row = name.endswith("embed_tokens.weight") | |
| c = (w32 @ r) if is_row else (r @ w32) | |
| edit = torch.outer(c, r) if is_row else torch.outer(r, c) | |
| tensors[name] = (w32 - edit).to(dtype=w.dtype) | |
| total_edited += 1 | |
| tmp_path = shard_path + ".tmp" | |
| save_file(tensors, tmp_path, metadata=metadata) | |
| os.replace(tmp_path, shard_path) | |
| with open(shard_done_marker, "w", encoding="utf-8") as mf: | |
| mf.write("ok\n") | |
| print(f" [{idx}/{len(by_shard)}] {shard}: abliterated {len(shard_names)} tensors in {time.time() - s_t0:.1f}s", flush=True) | |
| with open(ABLIT_MARKER, "w", encoding="utf-8") as mf: | |
| mf.write(f"edited={total_edited} time={time.time() - t0:.1f}s\n") | |
| print(f" -- Abliteration complete! Edited {total_edited} tensors in {time.time() - t0:.1f}s.", flush=True) | |
| def write_launchers(): | |
| think_bat = r"D:\Documentos\LLama\Launch Server - Swift 1.5 Qwen 3.8 27B Uncensored (THINKING - Official Guide).bat" | |
| with open(think_bat, "w", encoding="utf-8") as f: | |
| f.write(f"""@echo off | |
| title Swift 1.5 Qwen 3.8 27B Uncensored [THINKING MODE - Official Guide] (EXL3 3.75bpw H5 + DFlash2) | |
| echo ========================================================================= | |
| echo Swift 1.5 Qwen 3.8 27B Uncensored (SC_3.75bpw_H5_V6) + DFlash2 | |
| echo - Mode: THINKING MODE (enable_thinking=true, effort=xhigh) | |
| echo - Guardrails: Budget 2048, Preserve History (--reasoning-budget 2048 --reasoning-preserve) | |
| echo - Official: temp=0.6 (Coding/DFlash2), top_p=0.95, top_k=20, min_p=0.0, presence_penalty=1.5 | |
| echo - Target: Swift-1.5-Qwen3.8-27B-Uncensored-EXL3-3.75bpw (14.52 GiB, 5-bit Head, 6-bit Vision) | |
| echo - Drafter: Qwen3.8-27B-DFlash2-EXL3-5.0bpw (1.47 GB) | |
| echo - Context: 131,072 tokens (-cs 131072, 4,4-bit Hadamard KV Cache) | |
| echo - API: http://127.0.0.1:8080/v1 | |
| echo ========================================================================= | |
| echo. | |
| "D:\\Documentos\\LLama\\exl3_env\\Scripts\\python.exe" "D:\\Documentos\\LLama\\exl3_openai_server.py" ^ | |
| -m "{OUT_DIR}" ^ | |
| -dm "D:\\Documentos\\LLama\\models\\Qwen3.8-27B-DFlash2-EXL3-5.0bpw" ^ | |
| -cs 131072 ^ | |
| -cq 4,4 ^ | |
| --mode thinking ^ | |
| --reasoning-effort xhigh ^ | |
| --reasoning-budget 2048 ^ | |
| --reasoning-budget-message " ... reasoning budget reached, finalize response now." ^ | |
| --reasoning-preserve ^ | |
| --temp 0.6 ^ | |
| --top-p 0.95 ^ | |
| --top-k 20 ^ | |
| --min-p 0.0 ^ | |
| --presence-penalty 1.5 ^ | |
| --repetition-penalty 1.0 ^ | |
| --host 127.0.0.1 ^ | |
| --port 8080 | |
| pause | |
| """) | |
| print(f" -- Created launcher: {think_bat}", flush=True) | |
| def main(): | |
| if os.path.isfile(ABLIT_MARKER): | |
| print(" -- [Step 1/3 & 2/3] BF16 download and abliteration already completed (.abliterated_done)!", flush=True) | |
| else: | |
| print("=========================================================================", flush=True) | |
| print(" [Step 1/3] Downloading ukisai/Swift-1.5-Qwen3.8-27b (BF16, 51.77 GiB)...", flush=True) | |
| print(f" Target Directory: {BF16_DIR}", flush=True) | |
| print("=========================================================================", flush=True) | |
| snapshot_download( | |
| repo_id="ukisai/Swift-1.5-Qwen3.8-27b", | |
| local_dir=BF16_DIR, | |
| ignore_patterns=["*.mp4", "*.png", "*.jpg", "benchmarks/*"], | |
| ) | |
| print("\n -- [Step 1/3] BF16 download complete and verified!", flush=True) | |
| run_abliteration_in_place(BF16_DIR) | |
| print("=========================================================================", flush=True) | |
| print(" [Step 3/3] Starting Self-Calibrated EXL3 Conversion (SC_3.75bpw_H5_V6)...", flush=True) | |
| print(f" Output Directory: {OUT_DIR}", flush=True) | |
| print("=========================================================================", flush=True) | |
| free_comfyui_vram_if_idle() | |
| job_ckpt = os.path.join(WORK_DIR, "ckpt", "job.json") | |
| if os.path.isfile(job_ckpt): | |
| print(f" -- Found existing checkpoint ({job_ckpt}), resuming job...", flush=True) | |
| cmd = [ | |
| sys.executable, | |
| "-u", | |
| CONVERT_PY, | |
| "-w", WORK_DIR, | |
| "-r", | |
| "-d", "0", | |
| ] | |
| else: | |
| cmd = [ | |
| sys.executable, | |
| "-u", | |
| CONVERT_PY, | |
| "-i", BF16_DIR, | |
| "-w", WORK_DIR, | |
| "-o", OUT_DIR, | |
| "-rcp", RECIPE, | |
| "-cd", CAL_DATA, | |
| "-vb", "6", | |
| "-mb", "4", | |
| "-d", "0", | |
| ] | |
| ret = subprocess.call(cmd) | |
| if ret == 0: | |
| write_launchers() | |
| print("=========================================================================", flush=True) | |
| print(" SUCCESS! Swift-1.5-Qwen3.8-27B-Uncensored-EXL3-3.75bpw is READY!", flush=True) | |
| print("=========================================================================", flush=True) | |
| sys.exit(ret) | |
| if __name__ == "__main__": | |
| main() | |