{ "format": "bit-jev-single-question-case-v1", "scope": "one fixed development request; timings exclude model loading", "workload": { "questions": 1, "input_tokens": 703, "options": 77 }, "hardware": { "cpu": "Intel Xeon Gold 6459C", "gpu": "NVIDIA GeForce RTX 5090" }, "measurements": [ { "path": "native_cpu_i2_s_8_threads", "mean_inference_ms": 3127.88, "repetitions": 3, "peak_process_rss_mib": 1622.74 }, { "path": "native_cpu_i2_s_16_threads", "mean_inference_ms": 1972.17, "repetitions": 3, "peak_process_rss_mib": 1624.52 }, { "path": "experimental_gpu_fp16_mixed_precision", "mean_inference_ms": 86.56, "repetitions": 5, "peak_gpu_allocation_mib": 4935.53 } ], "limitations": [ "CPU uses the I2_S native runner while GPU uses an experimental FP16 mixed-precision path.", "This is one request with few repetitions, not a general throughput or accuracy claim.", "CPU process RSS and GPU allocation are different memory measurements.", "Generated tokens per second is not applicable: the model scores explicit options and emits structured decisions." ] }