Machine-Generated Text Detection on TruthfulQA
98.38TPR@FPR-1% (ChatGLM)FourierGPT-M
Evaluation Results
| Method | Links | ||||||||
|---|---|---|---|---|---|---|---|---|---|
| FourierGPT-Menhancement=Markov-informed calibration2026.05 | 98.38 | — | — | 86.86 | 97.71 | 84.57 | 96.14 | 92.73 | |
| Binoculars-Multenhancement=Multi-level Contextual Token Relation Modeling2026.05 | 98.34 | — | — | 87.35 | 97.38 | 86.88 | 96.5 | 93.29 | |
| FourierGPT-Multenhancement=Multi-level Contextual Token Relation Modeling2026.05 | 98.28 | — | — | 86.52 | 97.56 | 85.46 | 96.48 | 92.86 | |
| DetectGPT-Multenhancement=Multi-level Contextual Token Relation Modeling2026.05 | 98.23 | — | — | 87.48 | 97.49 | 85.32 | 96.43 | 92.99 | |
| FourierGPTenhancement=none2026.05 | 98.15 | — | — | 86.75 | 97.79 | 84.39 | 96.3 | 92.68 | |
| AdaGPT-Multenhancement=Multi-level Contextual Token Relation Modeling2026.05 | 98.05 | — | — | 85.6 | 96.92 | 82.24 | 95.3 | 91.62 | |
| Entropy-Multenhancement=Multi-level Contextual Token Relation Modeling2026.05 | 97.96 | — | — | 85.84 | 97.5 | 86.65 | 96.32 | 92.86 | |
| DetectLLM-Multenhancement=Multi-level Contextual Token Relation Modeling2026.05 | 97.79 | — | — | 86.55 | 97.17 | 84.59 | 95.43 | 92.3 | |
| Log-Rank-Multenhancement=Multi-level Contextual Token Relation Modeling2026.05 | 97.74 | — | — | 84.18 | 96.88 | 84.92 | 95.72 | 91.89 | |
| Likelihood-Multenhancement=Multi-level Contextual Token Relation Modeling2026.05 | 97.72 | — | — | 84.94 | 96.98 | 87.29 | 96.2 | 92.63 | |
| Log-Rank-Menhancement=Markov-informed calibration2026.05 | 97.56 | — | — | 82.14 | 96.36 | 84.16 | 95.1 | 91.06 | |
| Entropy-Menhancement=Markov-informed calibration2026.05 | 97.45 | — | — | 84.57 | 96.83 | 85.48 | 95.42 | 91.95 | |
| Likelihood-Menhancement=Markov-informed calibration2026.05 | 97.14 | — | — | 83.7 | 94.2 | 86.61 | 94.37 | 91.2 | |
| Likelihoodenhancement=none2026.05 | 97.08 | — | — | 79.28 | 95.65 | 84.67 | 94.17 | 90.17 | |
| Log-Rankenhancement=none2026.05 | 97.03 | — | — | 78.07 | 95.11 | 82.55 | 94.09 | 89.37 | |
| AdaGPTenhancement=none2026.05 | 96.67 | — | — | 80.57 | 92.46 | 78.29 | 92.96 | 88.19 | |
| FastGPT-Multenhancement=Multi-level Contextual Token Relation Modeling2026.05 | 96.65 | — | — | 82.38 | 96.97 | 85.38 | 94.72 | 91.22 | |
| FourierGPT-MultTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Multi-level Contextual Token Relation Modeling2026.05 | 96.56 | — | — | 69.51 | 93.14 | 56.79 | 89.52 | 81.1 | |
| FourierGPT-MTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Markov-informed calibration2026.05 | 96.45 | — | — | 70.1 | 92.69 | 56.28 | 89.18 | 80.94 | |
| FourierGPTTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Base2026.05 | 96.28 | — | — | 69.51 | 92.24 | 56.33 | 89.12 | 80.7 | |
| FastGPTenhancement=none2026.05 | 96.26 | — | — | 81.63 | 94.38 | 82.61 | 91.84 | 89.34 | |
| Entropyenhancement=none2026.05 | 96.02 | — | — | 80.5 | 94.04 | 79.29 | 91.3 | 88.23 | |
| AdaGPT-Menhancement=Markov-informed calibration2026.05 | 95.99 | — | — | 75.43 | 89.7 | 62.15 | 89.02 | 82.46 | |
| Entropy-MultTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Multi-level Contextual Token Relation Modeling2026.05 | 94.84 | — | — | 58.4 | 86.12 | 46.82 | 79.43 | 73.12 | |
| Binocularsenhancement=none2026.05 | 93.93 | — | — | 75 | 90.4 | 85.19 | 90.35 | 86.97 | |
| FastGPT-Menhancement=Markov-informed calibration2026.05 | 93.34 | — | — | 71.29 | 93.12 | 79.8 | 88.5 | 85.21 | |
| DetectGPTenhancement=none2026.05 | 92.84 | — | — | 72.25 | 92.84 | 75.73 | 86.7 | 84.07 | |
| AdaGPT-MultTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Multi-level Contextual Token Relation Modeling2026.05 | 92.78 | — | — | 54.24 | 79.21 | 40.74 | 76.54 | 68.7 | |
| Likelihood-MultTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Multi-level Contextual Token Relation Modeling2026.05 | 90.61 | — | — | 57.47 | 77.62 | 45.1 | 75.69 | 69.3 | |
| Log-Rank-MTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Markov-informed calibration2026.05 | 90.61 | — | — | 44.48 | 83.57 | 47.74 | 79.26 | 69.13 | |
| Log-Rank-MultTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Multi-level Contextual Token Relation Modeling2026.05 | 89.46 | — | — | 53.65 | 84.53 | 46.48 | 81.47 | 71.12 | |
| FastGPT-MultTraining Source=GPT4 texts, Proxy Model=GPT-2-XL, Enhancement=Multi-level Contextual Token Relation Modeling2026.05 | 89.46 | — | — | 45.47 | 83.97 | 43.15 | 74.62 | 67.33 | |
| AdaGPT-MTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Markov-informed calibration2026.05 | 88.72 | — | — | 42.95 | 64.76 | 30.2 | 70.54 | 59.43 | |
| DetectGPT-MultTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Multi-level Contextual Token Relation Modeling2026.05 | 88.66 | — | — | 55.82 | 86.69 | 41.09 | 79.6 | 70.37 | |
| Binoculars-MultTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Multi-level Contextual Token Relation Modeling2026.05 | 87.8 | — | — | 53.89 | 70.99 | 44.7 | 74.79 | 66.43 | |
| DetectGPT-Menhancement=Markov-informed calibration2026.05 | 87.21 | — | — | 80.43 | 81.58 | 75.07 | 82.52 | 81.36 | |
| Binoculars-Menhancement=Markov-informed calibration2026.05 | 86.82 | — | — | 74.48 | 77.21 | 75.15 | 80.92 | 78.92 | |
| Entropy-MTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Markov-informed calibration2026.05 | 84.65 | — | — | 52.7 | 78.87 | 39.94 | 69.52 | 65.13 | |
| FastGPT-MTraining Source=GPT4 texts, Proxy Model=GPT-2-XL, Enhancement=Markov-informed calibration2026.05 | 83.27 | — | — | 30.31 | 73.94 | 35.76 | 63.46 | 57.35 | |
| Log-RankTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Base2026.05 | 83.22 | — | — | 16.45 | 54.5 | 40.17 | 59.26 | 50.72 | |
| Likelihood-MTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Markov-informed calibration2026.05 | 75.91 | — | — | 52.17 | 42.04 | 29.17 | 54.79 | 50.82 | |
| DetectLLM-MultTraining Source=GPT4 texts, Proxy Model=GPT-2-XL, Enhancement=Multi-level Contextual Token Relation Modeling2026.05 | 75.31 | — | — | 40.68 | 63.68 | 40.52 | 58.81 | 55.8 | |
| LikelihoodTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Base2026.05 | 73.54 | — | — | 17.74 | 52.01 | 38.45 | 50.14 | 46.38 | |
| AdaGPTTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Base2026.05 | 71.99 | — | — | 36.02 | 47.31 | 19.54 | 50.76 | 45.12 | |
| DetectLLMenhancement=none2026.05 | 70.21 | — | — | 63.13 | 72.73 | 61.57 | 70.07 | 67.54 | |
| FastGPTTraining Source=GPT4 texts, Proxy Model=GPT-2-XL, Enhancement=Base2026.05 | 66.61 | — | — | 20.63 | 52.58 | 21.2 | 49.92 | 42.19 | |
| DetectLLM-Menhancement=Markov-informed calibration2026.05 | 63.21 | — | — | 59.95 | 63.91 | 60.29 | 64.37 | 62.35 | |
| DetectGPTTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Base2026.05 | 60.82 | — | — | 13.75 | 52.41 | 21.66 | 39.49 | 37.63 | |
| EntropyTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Base2026.05 | 60.48 | — | — | 18.38 | 38.02 | 24.3 | 32.29 | 34.69 | |
| Binoculars-MTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Markov-informed calibration2026.05 | 46.65 | — | — | 12.86 | 10.2 | 17.59 | 24.08 | 22.28 | |
| BinocularsTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Base2026.05 | 8.08 | — | — | 1.88 | 3.68 | 6.02 | 6.86 | 5.3 | |
| DetectLLMTraining Source=GPT4 texts, Proxy Model=GPT-2-XL, Enhancement=Base2026.05 | 1.2 | — | — | 1.7 | 0.57 | 0.86 | 1.08 | 1.08 | |
| DetectLLM-MTraining Source=GPT4 texts, Proxy Model=GPT-2-XL, Enhancement=Markov-informed calibration2026.05 | 0.51 | — | — | 1.65 | 0.28 | 0.4 | 0.62 | 0.69 | |
| DetectGPT-MTraining Source=GPT4 texts, Proxy Model=GPT-2, Enhancement=Markov-informed calibration2026.05 | 0.29 | — | — | 1.06 | 0.34 | 1.03 | 0.06 | 0.55 | |
| DetectGPTGenerator Model=GPT4, Training Source=GPT42026.02 | — | 21.66 | 75.73 | — | — | — | — | — | |
| DetectGPT-EGenerator Model=GPT4, Training Source=GPT42026.02 | — | 46.3 | 70.03 | — | — | — | — | — | |
| DetectGPT-EGenerator Model=ChatGPT-turbo, Training Source=GPT42026.02 | — | 82.31 | 90.38 | — | — | — | — | — | |
| DetectGPT-EGenerator Model=ChatGLM, Training Source=GPT42026.02 | — | 94.85 | 97.23 | — | — | — | — | — | |
| DetectGPT-EGenerator Model=Dolly, Training Source=GPT42026.02 | — | 59.52 | 78.97 | — | — | — | — | — | |
| DetectGPT-EGenerator Model=ChatGPT, Training Source=GPT42026.02 | — | 86.91 | 93.82 | — | — | — | — | — | |
| DetectGPT-EGenerator Model=StableLM, Training Source=GPT42026.02 | — | 84.14 | 91.37 | — | — | — | — | — | |
| DetectGPT-EGenerator Model=Avg., Training Source=GPT42026.02 | — | 75.67 | 86.96 | — | — | — | — | — | |
| DNAGPTGenerator Model=GPT4, Training Source=GPT42026.02 | — | 10.37 | 77.49 | — | — | — | — | — | |
| DNAGPT-EGenerator Model=GPT4, Training Source=GPT42026.02 | — | 8.08 | 83.92 | — | — | — | — | — | |
| DNAGPT-EGenerator Model=ChatGPT-turbo, Training Source=GPT42026.02 | — | 15.68 | 90.18 | — | — | — | — | — | |
| DNAGPT-EGenerator Model=ChatGLM, Training Source=GPT42026.02 | — | 20.39 | 92.96 | — | — | — | — | — | |
| DNAGPT-EGenerator Model=Dolly, Training Source=GPT42026.02 | — | 11.52 | 79.81 | — | — | — | — | — | |
| DNAGPT-EGenerator Model=ChatGPT, Training Source=GPT42026.02 | — | 21.47 | 92.16 | — | — | — | — | — | |
| DNAGPT-EGenerator Model=StableLM, Training Source=GPT42026.02 | — | 13.99 | 90.9 | — | — | — | — | — | |
| DNAGPT-EGenerator Model=Avg., Training Source=GPT42026.02 | — | 15.19 | 88.32 | — | — | — | — | — | |
| EntropyGenerator Model=GPT4, Training Source=GPT42026.02 | — | 24.3 | 79.29 | — | — | — | — | — | |
| Entropy-EGenerator Model=GPT4, Training Source=GPT42026.02 | — | 39.94 | 85.48 | — | — | — | — | — | |
| Entropy-EGenerator Model=ChatGPT-turbo, Training Source=GPT42026.02 | — | 78.44 | 95.17 | — | — | — | — | — | |
| Entropy-EGenerator Model=ChatGLM, Training Source=GPT42026.02 | — | 84.65 | 97.45 | — | — | — | — | — | |
| Entropy-EGenerator Model=Dolly, Training Source=GPT42026.02 | — | 52.7 | 84.57 | — | — | — | — | — | |
| Entropy-EGenerator Model=ChatGPT, Training Source=GPT42026.02 | — | 78.87 | 96.83 | — | — | — | — | — | |
| Entropy-EGenerator Model=StableLM, Training Source=GPT42026.02 | — | 69.52 | 95.42 | — | — | — | — | — | |
| Entropy-EGenerator Model=Avg., Training Source=GPT42026.02 | — | 67.35 | 92.49 | — | — | — | — | — | |
| FastGPTGenerator Model=GPT4, Training Source=GPT42026.02 | — | 21.2 | 82.61 | — | — | — | — | — | |
| FastGPT-EGenerator Model=GPT4, Training Source=GPT42026.02 | — | 37.25 | 82.8 | — | — | — | — | — | |
| FastGPT-EGenerator Model=ChatGPT-turbo, Training Source=GPT42026.02 | — | 74.64 | 94.26 | — | — | — | — | — | |
| FastGPT-EGenerator Model=ChatGLM, Training Source=GPT42026.02 | — | 89.8 | 97.68 | — | — | — | — | — | |
| FastGPT-EGenerator Model=Dolly, Training Source=GPT42026.02 | — | 49.06 | 84.02 | — | — | — | — | — | |
| FastGPT-EGenerator Model=ChatGPT, Training Source=GPT42026.02 | — | 83.8 | 96.47 | — | — | — | — | — | |
| FastGPT-EGenerator Model=StableLM, Training Source=GPT42026.02 | — | 76.32 | 94.84 | — | — | — | — | — | |
| FastGPT-EGenerator Model=Avg., Training Source=GPT42026.02 | — | 68.48 | 91.68 | — | — | — | — | — | |
| LikelihoodGenerator Model=GPT4, Training Source=GPT42026.02 | — | 38.45 | 84.67 | — | — | — | — | — | |
| Likelihood-EGenerator Model=GPT4, Training Source=GPT42026.02 | — | 29.17 | 86.61 | — | — | — | — | — | |
| Likelihood-EGenerator Model=ChatGPT-turbo, Training Source=GPT42026.02 | — | 56.89 | 92.72 | — | — | — | — | — | |
| Likelihood-EGenerator Model=ChatGLM, Training Source=GPT42026.02 | — | 75.91 | 97.14 | — | — | — | — | — | |
| Likelihood-EGenerator Model=Dolly, Training Source=GPT42026.02 | — | 52.17 | 83.7 | — | — | — | — | — | |
| Likelihood-EGenerator Model=ChatGPT, Training Source=GPT42026.02 | — | 42.04 | 94.2 | — | — | — | — | — | |
| Likelihood-EGenerator Model=StableLM, Training Source=GPT42026.02 | — | 54.79 | 94.37 | — | — | — | — | — | |
| Likelihood-EGenerator Model=Avg., Training Source=GPT42026.02 | — | 51.83 | 91.46 | — | — | — | — | — | |
| Log-RankGenerator Model=GPT4, Training Source=GPT42026.02 | — | 40.17 | 82.55 | — | — | — | — | — | |
| Log-Rank-EGenerator Model=GPT4, Training Source=GPT42026.02 | — | 47.74 | 84.16 | — | — | — | — | — | |
| Log-Rank-EGenerator Model=ChatGPT-turbo, Training Source=GPT42026.02 | — | 75.39 | 93.54 | — | — | — | — | — | |
| Log-Rank-EGenerator Model=ChatGLM, Training Source=GPT42026.02 | — | 90.55 | 97.56 | — | — | — | — | — | |
| Log-Rank-EGenerator Model=Dolly, Training Source=GPT42026.02 | — | 44.48 | 82.14 | — | — | — | — | — | |
| Log-Rank-EGenerator Model=ChatGPT, Training Source=GPT42026.02 | — | 83.57 | 96.36 | — | — | — | — | — |