Machine Unlearning on TOFU (1%)
0.0002Forget Quality (FQ)ECO
Evaluation Results
| Method | Links | |||||||
|---|---|---|---|---|---|---|---|---|
| ECOvariant=Sign-Flip2026.04 | 0.0002 | — | — | — | 0 | — | — | |
| Target LLM2024.06 | 0.001 | 95.2 | 0.62 | 98.2 | — | — | — | |
| OriginalCondition=Best-epoch performance, Evaluation=Averaged over five seeds2025.10 | 0.001 | — | 0.62 | — | — | — | — | |
| Original2025.10 | 0.001 | — | 0.62 | — | — | — | — | |
| Original LLM2025.10 | 0.001 | — | 0.62 | — | — | — | — | |
| Original2026.04 | 0.003 | — | — | — | 0 | — | — | |
| Original2026.04 | 0.003 | — | 0 | — | — | — | — | |
| Grad Ascent2026.04 | 0.0068 | — | — | — | -0.0233 | — | — | |
| KL Min2026.04 | 0.0068 | — | — | — | -0.0221 | — | — | |
| Prompt2026.04 | 0.0068 | — | — | — | -0.0628 | — | — | |
| GA2026.04 | 0.0068 | — | 0.0233 | — | — | — | — | |
| Grad Diff2026.04 | 0.0143 | — | — | — | -0.0198 | — | — | |
| Grad Diff2026.04 | 0.0143 | — | 0.0198 | — | — | — | — | |
| GA+KLCondition=Best-epoch performance, Evaluation=Averaged over five seeds2025.10 | 0.05 | — | 0.56 | — | — | — | — | |
| Pref Opt2026.04 | 0.0971 | — | — | — | -0.0021 | — | — | |
| Pref Opt2026.04 | 0.0971 | — | 0.0021 | — | — | — | — | |
| Offset-DPO+KLuse_offset_objective=true, loss=DPO+KL2024.06 | 0.13 | 3.8 | 0.12 | 19.1 | — | — | — | |
| DPO+GDforget_loss=DPO, retain_loss=GD2024.06 | 0.25 | 4.08 | 0.58 | 56.5 | — | — | — | |
| DPO+KLforget_loss=DPO, retain_loss=KL2024.06 | 0.26 | 4.18 | 0.58 | 55.6 | — | — | — | |
| GA+GDforget_loss=GA, retain_loss=GD2024.06 | 0.27 | 30.5 | 0.53 | 58.9 | — | — | — | |
| DPO2024.06 | 0.27 | 4.09 | 0.58 | 55.2 | — | — | — | |
| Offset-GA+KLuse_offset_objective=true, loss=GA+KL2024.06 | 0.27 | 44.7 | 0.52 | 45.8 | — | — | — | |
| DPO+GDCondition=Best-epoch performance, Evaluation=Averaged over five seeds2025.10 | 0.27 | — | 0.58 | — | — | — | — | |
| GA+GD2025.10 | 0.27 | — | 0.53 | — | — | — | — | |
| DPO+GD2025.10 | 0.27 | — | 0.58 | — | — | — | — | |
| GA+KL2025.10 | 0.31 | — | 0.53 | — | — | — | — | |
| GA2024.06 | 0.4 | 34.4 | 0.52 | 59.6 | — | — | — | |
| GA+KLforget_loss=GA, retain_loss=KL2024.06 | 0.4 | 35.2 | 0.53 | 59.9 | — | — | — | |
| GA+GDCondition=Best-epoch performance, Evaluation=Averaged over five seeds2025.10 | 0.4 | — | 0.53 | — | — | — | — | |
| GA2025.10 | 0.4 | — | 0.52 | — | — | — | — | |
| NPO-KL2026.04 | 0.4046 | — | — | — | -0.1703 | — | — | |
| RELOAD2026.04 | 0.4046 | — | — | — | 0.0748 | — | — | |
| RELOAD2026.04 | 0.4046 | — | 0.0748 | — | — | — | — | |
| Offset-NPO+KLuse_offset_objective=true, loss=NPO+KL2024.06 | 0.41 | 31.4 | 0.43 | 34.5 | — | — | — | |
| NPO+KLforget_loss=NPO, retain_loss=KL2024.06 | 0.52 | 33.7 | 0.54 | 58.7 | — | — | — | |
| GACondition=Best-epoch performance, Evaluation=Averaged over five seeds2025.10 | 0.57 | — | 0.55 | — | — | — | — | |
| NPO-RT2026.04 | 0.5786 | — | — | — | -0.1361 | — | — | |
| NPO-RT2026.04 | 0.5786 | — | 0.1361 | — | — | — | — | |
| NPO+GDforget_loss=NPO, retain_loss=GD2024.06 | 0.58 | 34.5 | 0.57 | 63.1 | — | — | — | |
| NPO2024.06 | 0.66 | 39.2 | 0.52 | 62.8 | — | — | — | |
| NPOCondition=Best-epoch performance, Evaluation=Averaged over five seeds2025.10 | 0.71 | — | 0.56 | — | — | — | — | |
| NPO+GDCondition=Best-epoch performance, Evaluation=Averaged over five seeds2025.10 | 0.71 | — | 0.58 | — | — | — | — | |
| NPO2025.10 | 0.71 | — | 0.56 | — | — | — | — | |
| NPO+GD2025.10 | 0.73 | — | 0.58 | — | — | — | — | |
| NPO2026.04 | 0.7659 | — | — | — | -0.1725 | — | — | |
| DiPO2025.10 | 0.89 | — | 0.58 | — | — | — | — | |
| ECOvariant=Rand Noise2026.04 | 0.9188 | — | — | — | 0 | — | — | |
| AltPO2025.10 | 0.92 | — | 0.55 | — | — | — | — | |
| ULD2024.06 | 0.99 | 40.7 | 0.62 | 98.3 | — | — | — | |
| ECOvariant=Zero-Out2026.04 | 0.99 | — | — | — | 0 | — | — | |
| ECOMode=Zero-Out2026.04 | 0.99 | — | 0 | — | — | — | — | |
| DiPOCondition=Best-epoch performance, Evaluation=Averaged over five seeds2025.10 | 0.99 | — | 0.59 | — | — | — | — | |
| ULD2025.10 | 0.99 | — | 0.62 | — | — | — | — | |
| DiPO2025.10 | 0.99 | — | 0.59 | — | — | — | — | |
| Retain LLM2024.06 | 1 | 37.6 | 0.62 | 98.5 | — | — | — | |
| Retain2026.04 | 1 | — | — | — | -0.0131 | — | — | |
| RetrainCondition=Best-epoch performance, Evaluation=Averaged over five seeds2025.10 | 1 | — | 0.62 | — | — | — | — | |
| Retrain2025.10 | 1 | — | 0.62 | — | — | — | — | |
| Retrain LLM2025.10 | 1 | — | 0.62 | — | — | — | — | |
| FinetunedBackbone=Llama-3-8B-Instruct2026.05 | — | 98 | — | 94.7 | — | — | — | |
| FinetunedBackbone=Llama-2-7B-chat-hf2026.05 | — | 95 | — | 95 | — | — | — | |
| GABackbone=Llama-3-8B-Instruct2026.05 | — | 13.8 | — | 15.6 | — | — | — | |
| GABackbone=Llama-2-7B-chat-hf2026.05 | — | 22 | — | 17.2 | — | — | — | |
| ICCU (end-to-end)Backbone=Llama-3-8B-Instruct2026.05 | — | 2.3 | — | 90.6 | — | 97.5 | 7.3 | |
| ICCU (end-to-end)Backbone=Llama-2-7B-chat-hf2026.05 | — | 2.5 | — | 89.1 | — | 97.5 | 8 | |
| ICCU (filter + generate)Backbone=Llama-3-8B-Instruct2026.05 | — | 1.2 | — | 91.4 | — | 97.5 | 7.5 | |
| ICCU (filter + generate)Backbone=Llama-2-7B-chat-hf2026.05 | — | 2.5 | — | 89.1 | — | 97.5 | 8 | |
| O^3Backbone=Llama-3-8B-Instruct2026.05 | — | 2 | — | 72.3 | — | — | — | |
| O^3Backbone=Llama-2-7B-chat-hf2026.05 | — | 3.6 | — | 79.8 | — | — | — | |
| PretrainedBackbone=Llama-3-8B-Instruct2026.05 | — | 18.8 | — | 15.7 | — | — | — | |
| PretrainedBackbone=Llama-2-7B-chat-hf2026.05 | — | 16.2 | — | 13.6 | — | — | — | |
| RMUBackbone=Llama-3-8B-Instruct2026.05 | — | 8 | — | 72.4 | — | — | — | |
| RMUBackbone=Llama-2-7B-chat-hf2026.05 | — | 7.8 | — | 69.6 | — | — | — |