tiktits's picture
Upload model card, configs, tokenizer, and abliteration/ reproduction kit
3439dd5 verified
Raw History Blame Contribute Delete
9.94 kB
import os
os.environ["PYTORCH_CUDA_ALLOC_CONF"] = "max_split_size_mb:128"
import sys
import json
import time
import shutil
import subprocess
import urllib.request
from pathlib import Path
import torch
from safetensors import safe_open
from safetensors.torch import save_file
from huggingface_hub import snapshot_download
BF16_DIR = r"D:\Documentos\LLama\models\Swift-1.5-Qwen3.8-27B-Uncensored-BF16"
WORK_DIR = r"D:\Documentos\LLama\work_swift15_3.75bpw"
OUT_DIR = r"D:\Documentos\LLama\models\Swift-1.5-Qwen3.8-27B-Uncensored-EXL3-3.75bpw"
RECIPE = r"D:\Documentos\LLama\cal_qwen38\recipe_3.75bpw_H5.yaml"
CAL_DATA = r"D:\Documentos\LLama\cal_qwen38\cal_trace.safetensors"
CONVERT_PY = r"D:\Documentos\LLama\convert.py"
ABLIT_DIR = r"D:\Documentos\LLama\abliteration_swift15\abliteration"
R_PT_PATH = os.path.join(ABLIT_DIR, "r.pt")
RECOVER_JSON_PATH = os.path.join(ABLIT_DIR, "recover.json")
ABLIT_MARKER = os.path.join(BF16_DIR, ".abliterated_done")
def free_comfyui_vram_if_idle():
"""If GPU VRAM is held by an idle ComfyUI instance on port 8188, ask it to unload cached models cleanly."""
try:
free_b, total_b = torch.cuda.mem_get_info(0)
free_gib = free_b / (1024 ** 3)
print(f" -- Current GPU VRAM free: {free_gib:.2f} GiB / {total_b / (1024**3):.2f} GiB", flush=True)
if free_gib >= 16.0:
return
print(" -- Checking if ComfyUI (127.0.0.1:8188) is idle to release cached VRAM...", flush=True)
while True:
try:
with urllib.request.urlopen("http://127.0.0.1:8188/queue", timeout=5) as resp:
q = json.loads(resp.read().decode("utf-8"))
running = q.get("queue_running", [])
pending = q.get("queue_pending", [])
if not running and not pending:
req = urllib.request.Request(
"http://127.0.0.1:8188/free",
data=json.dumps({"unload_models": True, "free_memory": True}).encode("utf-8"),
headers={"Content-Type": "application/json"},
method="POST",
)
urllib.request.urlopen(req, timeout=10)
time.sleep(3)
free_b2, _ = torch.cuda.mem_get_info(0)
print(f" -- ComfyUI unloaded idle models. Free GPU VRAM is now {free_b2 / (1024**3):.2f} GiB!", flush=True)
break
else:
print(" -- ComfyUI is actively generating a job; waiting 15s before checking again...", flush=True)
time.sleep(15)
except Exception:
break
except Exception as e:
print(f" -- Note during VRAM check: {e}", flush=True)
def run_abliteration_in_place(ckpt_dir: str):
"""Apply ajgazin/orcarouter rank-1 refusal orthogonalization (131 tensors) in float32 on CPU."""
if os.path.isfile(ABLIT_MARKER):
print(" -- [Step 2/3] Refusal abliteration (.abliterated_done) already applied! Skipping.", flush=True)
return
print("=========================================================================", flush=True)
print(" [Step 2/3] Applying Single-Direction Refusal Abliteration (131 tensors)", flush=True)
print(" W' = W - r (r^T W) | E' = E - (E r) r^T (float32 -> BF16)", flush=True)
print("=========================================================================", flush=True)
r = torch.load(R_PT_PATH, map_location="cpu")["r"].to(dtype=torch.float32)
r = r / r.norm()
with open(RECOVER_JSON_PATH, "r", encoding="utf-8") as f:
changed_names = [e["name"] for e in json.load(f)["changed"]]
index_file = os.path.join(ckpt_dir, "model.safetensors.index.json")
with open(index_file, "r", encoding="utf-8") as f:
wmap = json.load(f)["weight_map"]
by_shard: dict[str, list[str]] = {}
for name in sorted(changed_names):
if name not in wmap:
raise RuntimeError(f"Expected tensor {name} not found in {index_file}")
by_shard.setdefault(wmap[name], []).append(name)
t0 = time.time()
total_edited = 0
for idx, (shard, shard_names) in enumerate(sorted(by_shard.items()), 1):
shard_path = os.path.join(ckpt_dir, shard)
shard_done_marker = shard_path + ".ablit_done"
if os.path.isfile(shard_done_marker):
print(f" [{idx}/{len(by_shard)}] {shard}: already abliterated ({len(shard_names)} tensors)", flush=True)
total_edited += len(shard_names)
continue
s_t0 = time.time()
tensors = {}
with safe_open(shard_path, framework="pt", device="cpu") as f:
metadata = f.metadata()
for k in f.keys():
tensors[k] = f.get_tensor(k)
for name in shard_names:
w = tensors[name]
w32 = w.to(dtype=torch.float32)
is_row = name.endswith("embed_tokens.weight")
c = (w32 @ r) if is_row else (r @ w32)
edit = torch.outer(c, r) if is_row else torch.outer(r, c)
tensors[name] = (w32 - edit).to(dtype=w.dtype)
total_edited += 1
tmp_path = shard_path + ".tmp"
save_file(tensors, tmp_path, metadata=metadata)
os.replace(tmp_path, shard_path)
with open(shard_done_marker, "w", encoding="utf-8") as mf:
mf.write("ok\n")
print(f" [{idx}/{len(by_shard)}] {shard}: abliterated {len(shard_names)} tensors in {time.time() - s_t0:.1f}s", flush=True)
with open(ABLIT_MARKER, "w", encoding="utf-8") as mf:
mf.write(f"edited={total_edited} time={time.time() - t0:.1f}s\n")
print(f" -- Abliteration complete! Edited {total_edited} tensors in {time.time() - t0:.1f}s.", flush=True)
def write_launchers():
think_bat = r"D:\Documentos\LLama\Launch Server - Swift 1.5 Qwen 3.8 27B Uncensored (THINKING - Official Guide).bat"
with open(think_bat, "w", encoding="utf-8") as f:
f.write(f"""@echo off
title Swift 1.5 Qwen 3.8 27B Uncensored [THINKING MODE - Official Guide] (EXL3 3.75bpw H5 + DFlash2)
echo =========================================================================
echo Swift 1.5 Qwen 3.8 27B Uncensored (SC_3.75bpw_H5_V6) + DFlash2
echo - Mode: THINKING MODE (enable_thinking=true, effort=xhigh)
echo - Guardrails: Budget 2048, Preserve History (--reasoning-budget 2048 --reasoning-preserve)
echo - Official: temp=0.6 (Coding/DFlash2), top_p=0.95, top_k=20, min_p=0.0, presence_penalty=1.5
echo - Target: Swift-1.5-Qwen3.8-27B-Uncensored-EXL3-3.75bpw (14.52 GiB, 5-bit Head, 6-bit Vision)
echo - Drafter: Qwen3.8-27B-DFlash2-EXL3-5.0bpw (1.47 GB)
echo - Context: 131,072 tokens (-cs 131072, 4,4-bit Hadamard KV Cache)
echo - API: http://127.0.0.1:8080/v1
echo =========================================================================
echo.
"D:\\Documentos\\LLama\\exl3_env\\Scripts\\python.exe" "D:\\Documentos\\LLama\\exl3_openai_server.py" ^
-m "{OUT_DIR}" ^
-dm "D:\\Documentos\\LLama\\models\\Qwen3.8-27B-DFlash2-EXL3-5.0bpw" ^
-cs 131072 ^
-cq 4,4 ^
--mode thinking ^
--reasoning-effort xhigh ^
--reasoning-budget 2048 ^
--reasoning-budget-message " ... reasoning budget reached, finalize response now." ^
--reasoning-preserve ^
--temp 0.6 ^
--top-p 0.95 ^
--top-k 20 ^
--min-p 0.0 ^
--presence-penalty 1.5 ^
--repetition-penalty 1.0 ^
--host 127.0.0.1 ^
--port 8080
pause
""")
print(f" -- Created launcher: {think_bat}", flush=True)
def main():
if os.path.isfile(ABLIT_MARKER):
print(" -- [Step 1/3 & 2/3] BF16 download and abliteration already completed (.abliterated_done)!", flush=True)
else:
print("=========================================================================", flush=True)
print(" [Step 1/3] Downloading ukisai/Swift-1.5-Qwen3.8-27b (BF16, 51.77 GiB)...", flush=True)
print(f" Target Directory: {BF16_DIR}", flush=True)
print("=========================================================================", flush=True)
snapshot_download(
repo_id="ukisai/Swift-1.5-Qwen3.8-27b",
local_dir=BF16_DIR,
ignore_patterns=["*.mp4", "*.png", "*.jpg", "benchmarks/*"],
)
print("\n -- [Step 1/3] BF16 download complete and verified!", flush=True)
run_abliteration_in_place(BF16_DIR)
print("=========================================================================", flush=True)
print(" [Step 3/3] Starting Self-Calibrated EXL3 Conversion (SC_3.75bpw_H5_V6)...", flush=True)
print(f" Output Directory: {OUT_DIR}", flush=True)
print("=========================================================================", flush=True)
free_comfyui_vram_if_idle()
job_ckpt = os.path.join(WORK_DIR, "ckpt", "job.json")
if os.path.isfile(job_ckpt):
print(f" -- Found existing checkpoint ({job_ckpt}), resuming job...", flush=True)
cmd = [
sys.executable,
"-u",
CONVERT_PY,
"-w", WORK_DIR,
"-r",
"-d", "0",
]
else:
cmd = [
sys.executable,
"-u",
CONVERT_PY,
"-i", BF16_DIR,
"-w", WORK_DIR,
"-o", OUT_DIR,
"-rcp", RECIPE,
"-cd", CAL_DATA,
"-vb", "6",
"-mb", "4",
"-d", "0",
]
ret = subprocess.call(cmd)
if ret == 0:
write_launchers()
print("=========================================================================", flush=True)
print(" SUCCESS! Swift-1.5-Qwen3.8-27B-Uncensored-EXL3-3.75bpw is READY!", flush=True)
print("=========================================================================", flush=True)
sys.exit(ret)
if __name__ == "__main__":
main()