model.gguf benchmark on an AMD's logo.EPYC

<- Runs

Prompt tokens

8,192

Generation tokens

2,048

Trials passed

2/2

Rejected

300.5 tok/s

1,300.0 tok/s

Peak memory

0.00/4 GB

Runs great

Trials

Decode / Prefill Speeds

Metadata

metadata.json
{
"runId": "run_48f2fbf4-e59a-4652-9c38-9e2e921a78bb",
"bundleId": "llamacpp-model.gguf-e32f56",
"status": "rejected",
"promptTokens": 8192,
"completionTokens": 2048,
"contextLength": 5120,
"harness": {
"version": "0.1.12",
"gitSha": "01f7616"
},
"runtime": {
"name": "llama.cpp",
"version": "b8240",
"buildFlags": "metal"
},
"model": {
"displayName": "model.gguf",
"format": "gguf",
"quant": null,
"architecture": null,
"source": null,
"fileSizeBytes": 2048,
"lab": null,
"quantizedBy": null
},
"device": {
"cpu": "AMD EPYC",
"cpuCores": 2,
"gpu": "None",
"gpuCores": 0,
"gpuCount": 0,
"ramGb": 4,
"osName": "Ubuntu 22.04 LTS",
"osVersion": "22.04"
},
"decodeTpsMean": 300.5,
"prefillTpsMean": 1300,
"ttftP50Ms": 3150.77,
"idleTpsMean": 3,
"peakRssMb": 3,
"trialsPassed": 2,
"trialsTotal": 2,
"runnabilityScore": 0.881,
"bundleSha256": "1ff7fca2c9783b057a06b59324e144ba734398b60eaa4514d3ae0cc3888ab36c",
"createdAt": "2026-09-03T22:28:41.516Z"
}