DeepSeek-V4 Technical Report Benchmarks
GPT Image 2

DeepSeek-V4 Technical Report Benchmarks

gpt-image-2-prompt-deepseek-v4-technical-r.txt
{
 "type": "photograph of a computer monitor displaying an academic technical report",
 "style": "slightly angled screen photo, visible moire pattern, LCD pixel grid, slight glare, LaTeX document formatting, serif fonts",
 "document_header": {
 "left": "4 Benchmark Evaluation",
 "right": "DeepSeek-V4 Technical Report\"
 },
 "introductory_text": "Paragraph summarizing comprehensive evaluation of DeepSeek-V4\ against GPT-5.3\, Claude Opus 4.6\, and Gemini 3.1 Pro Preview\.",
 "visualizations": {
 "legend": "5 items with color codes: dark blue, grey, light grey, blue striped, light blue",
 "bar_charts": {
 "count": 6,
 "labels": [
 "MMLU-Pro (EM)",
 "GPQA-Diamond (Pass@1)",
 "AIME 2025 (Pass@1)",
 "LiveCodeBench (Pass@1-COT)",
 "SWE-bench Verified (Resolved)",
 "Tau-bench (Average)"
 ]
 },
 "caption": "Figure 1 | Performance comparison on core benchmarks. DeepSeek-V4 achieves state-of-the-art results across the majority of benchmarks."
 },
 "data_table": {
 "columns": [
 "Benchmark",
 "DeepSeek-V4\",
 "GPT-5.3\",
 "Claude Opus 4.6\",
 "Gemini 3.1 Pro Preview\",
 "GPT-4.1"
 ],
 "categories": {
 "count": 4,
 "rows": [
 {"label": "General", "icon": "globe/network", "sub_items": 3},
 {"label": "Reasoning & Math", "icon": "calculator/clipboard", "sub_items": 3},
 {"label": "Code", "icon": "code brackets", "sub_items": 3},
 {"label": "Agent", "icon": "robot face", "sub_items": 3}
 ]
 }
 }
}