#!/usr/bin/env python3 """Is orcarouter's MTP-head edit a single-direction (Robinson-style) projection? Robinson orthogonalizes a residual WRITER as W' = (I - d d^T) W, so delta = W' - W = -d d^T W is EXACTLY RANK 1 with left singular vector d. Two independent predictions follow, both checkable from the weights alone: 1. sigma_2/sigma_1 ~ 0 for each edited tensor 2. the d recovered from o_proj and from down_proj must AGREE (|cos| ~ 1), because Robinson uses ONE shared direction Then sink-screen the recovered d: Heretic's was 6.18% concentrated in dim 3994, which is what made an in-band graft unsafe on that trunk. Ours want < 1%. """ import json, torch from pathlib import Path from safetensors import safe_open CAND = Path("/tank/aimodels/qwen38-27b-orcarouter-bf16") REF = Path("/tank/aimodels/qwen38-27b-uncensored-bf16") KEYS = ["mtp.layers.0.self_attn.o_proj.weight", "mtp.layers.0.mlp.down_proj.weight"] SINK_DIM = 3994 def get(d: Path, key: str): idx = json.loads((d / "model.safetensors.index.json").read_text())["weight_map"] with safe_open(d / idx[key], framework="pt") as f: return f.get_tensor(key) dirs = {} for k in KEYS: delta = (get(CAND, k).float() - get(REF, k).float()) U, S, Vh = torch.linalg.svd(delta, full_matrices=False) ratio = (S[1] / S[0]).item() d = U[:, 0] # residual-space direction (dim 5120) dirs[k] = d sink = (d[SINK_DIM] ** 2).item() / (d @ d).item() print(f" {k.split('.',2)[2]:<28} shape={tuple(delta.shape)}") print(f" sigma2/sigma1 = {ratio:.6f} <- rank-1 if ~0") print(f" ||delta||/||W|| = {(delta.norm()/get(REF,k).float().norm()).item():.5f}") print(f" sink energy dim {SINK_DIM} = {sink*100:.4f}% <- want <1%, Heretic's was 6.18%") a, b = dirs[KEYS[0]], dirs[KEYS[1]] cos = torch.abs(a @ b / (a.norm() * b.norm())).item() print(f"\n |cos| between the two recovered directions = {cos:.4f} <- ~1 means ONE shared direction") top = torch.topk(a.abs(), 5) print(f" top-5 |d| coords (o_proj): {[(int(i), round(float(v),4)) for i, v in zip(top.indices, top.values)]}")