[ { "50.00 percentile latency (ns)": 403763330804, "90.00 percentile latency (ns)": 727220976676, "95.00 percentile latency (ns)": 767584600949, "97.00 percentile latency (ns)": 783758163328, "99.00 percentile latency (ns)": 799804752661, "99.90 percentile latency (ns)": 807326018896, "Max latency (ns)": 807905874878, "Mean latency (ns)": 403871265653, "Min duration satisfied": "Yes", "Min latency (ns)": 59688159, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 168.287, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIE-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 4, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7803, "characteristics.mean": 0.85443, "characteristics.samples_per_second": 168.287, "characteristics.samples_per_second.normalized_per_core": 42.07175, "characteristics.samples_per_second.normalized_per_processor": 42.07175, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9133, "ck_system": "SYS-120GQ-TNRT_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.4", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 18, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Gold 6354", "host_processors_per_node": 2, "host_storage_capacity": "3.5 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 4, "normalize_processors": 4, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/results/SYS-120GQ-TNRT_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.1", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.4, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "N", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 135960, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/SYS-120GQ-TNRT_TRT", "system_name": "Supermicro SYS-120GQ-TNRT (4x A100-PCIe-80GB, TensorRT)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 206, "task": "image segmentation", "task2": "image segmentation", "total_cores": 36, "uid": "24bc48aeae422f22", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 403763330804, "90.00 percentile latency (ns)": 727220976676, "95.00 percentile latency (ns)": 767584600949, "97.00 percentile latency (ns)": 783758163328, "99.00 percentile latency (ns)": 799804752661, "99.90 percentile latency (ns)": 807326018896, "Max latency (ns)": 807905874878, "Mean latency (ns)": 403871265653, "Min duration satisfied": "Yes", "Min latency (ns)": 59688159, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 168.287, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIE-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 4, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7803, "characteristics.mean": 0.85443, "characteristics.samples_per_second": 168.287, "characteristics.samples_per_second.normalized_per_core": 42.07175, "characteristics.samples_per_second.normalized_per_processor": 42.07175, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9133, "ck_system": "SYS-120GQ-TNRT_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.4", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 18, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Gold 6354", "host_processors_per_node": 2, "host_storage_capacity": "3.5 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 4, "normalize_processors": 4, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/results/SYS-120GQ-TNRT_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.1", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.4, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "N", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 135960, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/SYS-120GQ-TNRT_TRT", "system_name": "Supermicro SYS-120GQ-TNRT (4x A100-PCIe-80GB, TensorRT)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 206, "task": "image segmentation", "task2": "image segmentation", "total_cores": 36, "uid": "e98e7adbb7d263ed", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 333924720346, "90.00 percentile latency (ns)": 612558353881, "95.00 percentile latency (ns)": 647638685070, "97.00 percentile latency (ns)": 661655263158, "99.00 percentile latency (ns)": 675692090915, "99.90 percentile latency (ns)": 681994464576, "Max latency (ns)": 682691300644, "Mean latency (ns)": 336344453035, "Min duration satisfied": "Yes", "Min latency (ns)": 103086144, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 222.355, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 8, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7803, "characteristics.mean": 0.85443, "characteristics.samples_per_second": 222.355, "characteristics.samples_per_second.normalized_per_core": 27.794375, "characteristics.samples_per_second.normalized_per_processor": 27.794375, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9133, "ck_system": "A30x8_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "7 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 8, "normalize_processors": 8, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/results/A30x8_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "N", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 151800, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "SUPERMICRO", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x8_TRT", "system_name": "Supermicro AS-4124GS-TNR (8x A30, TensorRT)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 230, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "09f8e4653a3cde75", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 333924720346, "90.00 percentile latency (ns)": 612558353881, "95.00 percentile latency (ns)": 647638685070, "97.00 percentile latency (ns)": 661655263158, "99.00 percentile latency (ns)": 675692090915, "99.90 percentile latency (ns)": 681994464576, "Max latency (ns)": 682691300644, "Mean latency (ns)": 336344453035, "Min duration satisfied": "Yes", "Min latency (ns)": 103086144, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 222.355, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 8, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7803, "characteristics.mean": 0.85443, "characteristics.samples_per_second": 222.355, "characteristics.samples_per_second.normalized_per_core": 27.794375, "characteristics.samples_per_second.normalized_per_processor": 27.794375, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9133, "ck_system": "A30x8_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "7 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 8, "normalize_processors": 8, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/results/A30x8_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "N", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 151800, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "SUPERMICRO", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x8_TRT", "system_name": "Supermicro AS-4124GS-TNR (8x A30, TensorRT)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 230, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "36aa4aae8dd2e179", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 351345007291, "90.00 percentile latency (ns)": 644277890144, "95.00 percentile latency (ns)": 681063902769, "97.00 percentile latency (ns)": 695793665913, "99.00 percentile latency (ns)": 710516567912, "99.90 percentile latency (ns)": 717140014640, "Max latency (ns)": 717851791220, "Mean latency (ns)": 353805356254, "Min duration satisfied": "Yes", "Min latency (ns)": 88601891, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 211.464, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 8, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7802, "characteristics.mean": 0.85437, "characteristics.samples_per_second": 211.464, "characteristics.samples_per_second.normalized_per_core": 26.433, "characteristics.samples_per_second.normalized_per_processor": 26.433, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9132, "ck_system": "A30x8_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "7 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 8, "normalize_processors": 8, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/results/A30x8_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "N", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 151800, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "SUPERMICRO", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x8_TRT_Triton", "system_name": "Supermicro AS-4124GS-TNR (8x A30, TensorRT, Triton)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 230, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "cbcd0029910a7337", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 351345007291, "90.00 percentile latency (ns)": 644277890144, "95.00 percentile latency (ns)": 681063902769, "97.00 percentile latency (ns)": 695793665913, "99.00 percentile latency (ns)": 710516567912, "99.90 percentile latency (ns)": 717140014640, "Max latency (ns)": 717851791220, "Mean latency (ns)": 353805356254, "Min duration satisfied": "Yes", "Min latency (ns)": 88601891, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 211.464, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 8, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7802, "characteristics.mean": 0.85437, "characteristics.samples_per_second": 211.464, "characteristics.samples_per_second.normalized_per_core": 26.433, "characteristics.samples_per_second.normalized_per_processor": 26.433, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9132, "ck_system": "A30x8_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "7 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 8, "normalize_processors": 8, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Supermicro/results/A30x8_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "N", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 151800, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "SUPERMICRO", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x8_TRT_Triton", "system_name": "Supermicro AS-4124GS-TNR (8x A30, TensorRT, Triton)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 230, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "18947e5aa02a718b", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 351645469648, "90.00 percentile latency (ns)": 631144354393, "95.00 percentile latency (ns)": 666187235593, "97.00 percentile latency (ns)": 680175811193, "99.00 percentile latency (ns)": 694184917584, "99.90 percentile latency (ns)": 700395888155, "Max latency (ns)": 701090141496, "Mean latency (ns)": 350988707174, "Min duration satisfied": "Yes", "Min latency (ns)": 65312762, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 451.868, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 8, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85387, "characteristics.samples_per_second": 451.868, "characteristics.samples_per_second.normalized_per_core": 56.4835, "characteristics.samples_per_second.normalized_per_processor": 56.4835, "characteristics.tumor core": 0.8684, "characteristics.whole tumor": 0.9132, "ck_system": "AS-4124GO-NART_A100-SXM4-40GBx8_TRT", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 8, "normalize_processors": 8, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/results/AS-4124GO-NART_A100-SXM4-40GBx8_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 316800, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AS-4124GO-NART_A100-SXM4-40GBx8_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 480, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "eee89b79dc1c9852", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 351645469648, "90.00 percentile latency (ns)": 631144354393, "95.00 percentile latency (ns)": 666187235593, "97.00 percentile latency (ns)": 680175811193, "99.00 percentile latency (ns)": 694184917584, "99.90 percentile latency (ns)": 700395888155, "Max latency (ns)": 701090141496, "Mean latency (ns)": 350988707174, "Min duration satisfied": "Yes", "Min latency (ns)": 65312762, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 451.868, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 8, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85387, "characteristics.samples_per_second": 451.868, "characteristics.samples_per_second.normalized_per_core": 56.4835, "characteristics.samples_per_second.normalized_per_processor": 56.4835, "characteristics.tumor core": 0.8684, "characteristics.whole tumor": 0.9132, "ck_system": "AS-4124GO-NART_A100-SXM4-40GBx8_TRT", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 8, "normalize_processors": 8, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/results/AS-4124GO-NART_A100-SXM4-40GBx8_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 316800, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AS-4124GO-NART_A100-SXM4-40GBx8_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 480, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "c494297bbf4c8a08", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 337404952564, "90.00 percentile latency (ns)": 607402981477, "95.00 percentile latency (ns)": 641175338955, "97.00 percentile latency (ns)": 654676325717, "99.00 percentile latency (ns)": 668175682679, "99.90 percentile latency (ns)": 674253662092, "Max latency (ns)": 674928484316, "Mean latency (ns)": 337424648668, "Min duration satisfied": "Yes", "Min latency (ns)": 49107490, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 234.692, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 4, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85387, "characteristics.samples_per_second": 234.692, "characteristics.samples_per_second.normalized_per_core": 58.673, "characteristics.samples_per_second.normalized_per_processor": 58.673, "characteristics.tumor core": 0.8684, "characteristics.whole tumor": 0.9132, "ck_system": "AS-2124GQ-NART_A100-SXM4-40GBx4_TRT", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "512 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 4, "normalize_processors": 4, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/results/AS-2124GQ-NART_A100-SXM4-40GBx4_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 158400, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AS-2124GQ-NART_A100-SXM4-40GBx4_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 240, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "0d5a78ea99b71fad", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 337404952564, "90.00 percentile latency (ns)": 607402981477, "95.00 percentile latency (ns)": 641175338955, "97.00 percentile latency (ns)": 654676325717, "99.00 percentile latency (ns)": 668175682679, "99.90 percentile latency (ns)": 674253662092, "Max latency (ns)": 674928484316, "Mean latency (ns)": 337424648668, "Min duration satisfied": "Yes", "Min latency (ns)": 49107490, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 234.692, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 4, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85387, "characteristics.samples_per_second": 234.692, "characteristics.samples_per_second.normalized_per_core": 58.673, "characteristics.samples_per_second.normalized_per_processor": 58.673, "characteristics.tumor core": 0.8684, "characteristics.whole tumor": 0.9132, "ck_system": "AS-2124GQ-NART_A100-SXM4-40GBx4_TRT", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "512 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 4, "normalize_processors": 4, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/results/AS-2124GQ-NART_A100-SXM4-40GBx4_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 158400, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AS-2124GQ-NART_A100-SXM4-40GBx4_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 240, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "0a36681c93408ab8", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 326261546364, "90.00 percentile latency (ns)": 587516531539, "95.00 percentile latency (ns)": 620168803185, "97.00 percentile latency (ns)": 633230015342, "99.00 percentile latency (ns)": 646305447564, "99.90 percentile latency (ns)": 652163745469, "Max latency (ns)": 652820887803, "Mean latency (ns)": 326308578604, "Min duration satisfied": "Yes", "Min latency (ns)": 45212396, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 242.639, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 4, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85387, "characteristics.samples_per_second": 242.639, "characteristics.samples_per_second.normalized_per_core": 60.65975, "characteristics.samples_per_second.normalized_per_processor": 60.65975, "characteristics.tumor core": 0.8684, "characteristics.whole tumor": 0.9132, "ck_system": "AS-2124GQ-NART_A100-SXM-80GBx4_TRT", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "3.5 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 4, "normalize_processors": 4, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/results/AS-2124GQ-NART_A100-SXM-80GBx4_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 158400, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AS-2124GQ-NART_A100-SXM-80GBx4_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 240, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "53fab634d7bad9b2", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 326261546364, "90.00 percentile latency (ns)": 587516531539, "95.00 percentile latency (ns)": 620168803185, "97.00 percentile latency (ns)": 633230015342, "99.00 percentile latency (ns)": 646305447564, "99.90 percentile latency (ns)": 652163745469, "Max latency (ns)": 652820887803, "Mean latency (ns)": 326308578604, "Min duration satisfied": "Yes", "Min latency (ns)": 45212396, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 242.639, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 4, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85387, "characteristics.samples_per_second": 242.639, "characteristics.samples_per_second.normalized_per_core": 60.65975, "characteristics.samples_per_second.normalized_per_processor": 60.65975, "characteristics.tumor core": 0.8684, "characteristics.whole tumor": 0.9132, "ck_system": "AS-2124GQ-NART_A100-SXM-80GBx4_TRT", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "3.5 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 4, "normalize_processors": 4, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/results/AS-2124GQ-NART_A100-SXM-80GBx4_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 158400, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AS-2124GQ-NART_A100-SXM-80GBx4_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 240, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "c1ef219e94b80853", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 358066992587, "90.00 percentile latency (ns)": 644681931024, "95.00 percentile latency (ns)": 680511740776, "97.00 percentile latency (ns)": 694833160930, "99.00 percentile latency (ns)": 709151872674, "99.90 percentile latency (ns)": 715602139345, "Max latency (ns)": 716306196541, "Mean latency (ns)": 358062684798, "Min duration satisfied": "Yes", "Min latency (ns)": 88347442, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 386.985, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 8, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85387, "characteristics.samples_per_second": 386.985, "characteristics.samples_per_second.normalized_per_core": 48.373125, "characteristics.samples_per_second.normalized_per_processor": 48.373125, "characteristics.tumor core": 0.8684, "characteristics.whole tumor": 0.9132, "ck_system": "A100-PCIex8_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 16, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7313", "host_processors_per_node": 2, "host_storage_capacity": "7 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 8, "normalize_processors": 8, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/results/A100-PCIex8_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 277200, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex8_TRT", "system_name": "Supermicro AS-4124GS-TNR (8x A100-PCIe)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 420, "task": "image segmentation", "task2": "image segmentation", "total_cores": 32, "uid": "3f98796d4514eb2b", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 358066992587, "90.00 percentile latency (ns)": 644681931024, "95.00 percentile latency (ns)": 680511740776, "97.00 percentile latency (ns)": 694833160930, "99.00 percentile latency (ns)": 709151872674, "99.90 percentile latency (ns)": 715602139345, "Max latency (ns)": 716306196541, "Mean latency (ns)": 358062684798, "Min duration satisfied": "Yes", "Min latency (ns)": 88347442, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 386.985, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 8, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85387, "characteristics.samples_per_second": 386.985, "characteristics.samples_per_second.normalized_per_core": 48.373125, "characteristics.samples_per_second.normalized_per_processor": 48.373125, "characteristics.tumor core": 0.8684, "characteristics.whole tumor": 0.9132, "ck_system": "A100-PCIex8_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 16, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7313", "host_processors_per_node": 2, "host_storage_capacity": "7 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 8, "normalize_processors": 8, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/results/A100-PCIex8_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 277200, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex8_TRT", "system_name": "Supermicro AS-4124GS-TNR (8x A100-PCIe)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 420, "task": "image segmentation", "task2": "image segmentation", "total_cores": 32, "uid": "7dfe0f0f0024412b", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 329055503485, "90.00 percentile latency (ns)": 592433355364, "95.00 percentile latency (ns)": 625357472555, "97.00 percentile latency (ns)": 638523800067, "99.00 percentile latency (ns)": 651692027216, "99.90 percentile latency (ns)": 657615172090, "Max latency (ns)": 658273517300, "Mean latency (ns)": 329066730025, "Min duration satisfied": "Yes", "Min latency (ns)": 50201688, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 481.259, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 8, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85387, "characteristics.samples_per_second": 481.259, "characteristics.samples_per_second.normalized_per_core": 60.157375, "characteristics.samples_per_second.normalized_per_processor": 60.157375, "characteristics.tumor core": 0.8684, "characteristics.whole tumor": 0.9132, "ck_system": "SYS-420GP-TNAR_A100-SXM-80GBx8_TRT", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "3.5 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "normalize_cores": 8, "normalize_processors": 8, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/results/SYS-420GP-TNAR_A100-SXM-80GBx8_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 316800, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "preview", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/SYS-420GP-TNAR_A100-SXM-80GBx8_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 480, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "0441fff742debecc", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 329055503485, "90.00 percentile latency (ns)": 592433355364, "95.00 percentile latency (ns)": 625357472555, "97.00 percentile latency (ns)": 638523800067, "99.00 percentile latency (ns)": 651692027216, "99.90 percentile latency (ns)": 657615172090, "Max latency (ns)": 658273517300, "Mean latency (ns)": 329066730025, "Min duration satisfied": "Yes", "Min latency (ns)": 50201688, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 481.259, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 8, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85387, "characteristics.samples_per_second": 481.259, "characteristics.samples_per_second.normalized_per_core": 60.157375, "characteristics.samples_per_second.normalized_per_processor": 60.157375, "characteristics.tumor core": 0.8684, "characteristics.whole tumor": 0.9132, "ck_system": "SYS-420GP-TNAR_A100-SXM-80GBx8_TRT", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "3.5 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "normalize_cores": 8, "normalize_processors": 8, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/Supermicro/results/SYS-420GP-TNAR_A100-SXM-80GBx8_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 316800, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "preview", "submitter": "Supermicro", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Supermicro", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/SYS-420GP-TNAR_A100-SXM-80GBx8_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "datacenter", "target_latency (ns)": 0, "target_qps": 480, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "9b13fc50c1c77f10", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" } ]