.eval_results/GLM-5.3-Flash.yaml
845 B · 29 lines · yaml Raw
1 - dataset:
2 id: harborframework/terminal-bench-2.1
3 task_id: terminalbench_2_1
4 value: 84.3
5 date: "2026-08-26"
6 source:
7 url: https://huggingface.co/zai-org/GLM-5.3-Flash
8 name: "GLM-5.3-Flash model card"
9
10 - dataset:
11 id: datacurve/deep-swe
12 task_id: deep_swe
13 value: 63.4
14 date: "2026-08-26"
15 source:
16 url: https://huggingface.co/zai-org/GLM-5.3-Flash
17 name: "GLM-5.3-Flash model card"
18 notes: "Reported as DeepSWE v1.1 on the model card, run via the mini-swe-agent harness with 400K context."
19
20 - dataset:
21 id: cais/hle
22 task_id: hle
23 value: 55.3
24 date: "2026-08-26"
25 source:
26 url: https://huggingface.co/zai-org/GLM-5.3-Flash
27 name: "GLM-5.3-Flash model card"
28 notes: "HLE with tools (full set) and a 300K-context management strategy, not the no-tools default; judged by GPT-5.6-luna (medium)."
29