.eval_results/GLM-5.3-Flash.yaml
| 1 | - dataset: |
| 2 | id: harborframework/terminal-bench-2.1 |
| 3 | task_id: terminalbench_2_1 |
| 4 | value: 84.3 |
| 5 | date: "2026-08-26" |
| 6 | source: |
| 7 | url: https://huggingface.co/zai-org/GLM-5.3-Flash |
| 8 | name: "GLM-5.3-Flash model card" |
| 9 | |
| 10 | - dataset: |
| 11 | id: datacurve/deep-swe |
| 12 | task_id: deep_swe |
| 13 | value: 63.4 |
| 14 | date: "2026-08-26" |
| 15 | source: |
| 16 | url: https://huggingface.co/zai-org/GLM-5.3-Flash |
| 17 | name: "GLM-5.3-Flash model card" |
| 18 | notes: "Reported as DeepSWE v1.1 on the model card, run via the mini-swe-agent harness with 400K context." |
| 19 | |
| 20 | - dataset: |
| 21 | id: cais/hle |
| 22 | task_id: hle |
| 23 | value: 55.3 |
| 24 | date: "2026-08-26" |
| 25 | source: |
| 26 | url: https://huggingface.co/zai-org/GLM-5.3-Flash |
| 27 | name: "GLM-5.3-Flash model card" |
| 28 | notes: "HLE with tools (full set) and a 300K-context management strategy, not the no-tools default; judged by GPT-5.6-luna (medium)." |
| 29 | |