[ { "50.00 percentile latency (ns)": 405560464520, "90.00 percentile latency (ns)": 730335789251, "95.00 percentile latency (ns)": 770870076569, "97.00 percentile latency (ns)": 787118447455, "99.00 percentile latency (ns)": 803364324604, "99.90 percentile latency (ns)": 810628605844, "Max latency (ns)": 811420346783, "Mean latency (ns)": 405657549159, "Min duration satisfied": "Yes", "Min latency (ns)": 134262467, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 30.2876, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85407, "characteristics.samples_per_second": 30.2876, "characteristics.samples_per_second.normalized_per_core": 30.2876, "characteristics.samples_per_second.normalized_per_processor": 30.2876, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.9142, "ck_system": "A30x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 30.74, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "36dc78db4401b771", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 405560464520, "90.00 percentile latency (ns)": 730335789251, "95.00 percentile latency (ns)": 770870076569, "97.00 percentile latency (ns)": 787118447455, "99.00 percentile latency (ns)": 803364324604, "99.90 percentile latency (ns)": 810628605844, "Max latency (ns)": 811420346783, "Mean latency (ns)": 405657549159, "Min duration satisfied": "Yes", "Min latency (ns)": 134262467, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 30.2876, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85407, "characteristics.samples_per_second": 30.2876, "characteristics.samples_per_second.normalized_per_core": 30.2876, "characteristics.samples_per_second.normalized_per_processor": 30.2876, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.9142, "ck_system": "A30x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 30.74, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "50a9fdb0c4af7acd", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 884117079, "90.00 percentile latency (ns)": 884570991, "90th percentile latency (ns)": 884570991, "95.00 percentile latency (ns)": 884734997, "97.00 percentile latency (ns)": 884882871, "99.00 percentile latency (ns)": 885074323, "99.90 percentile latency (ns)": 885446839, "Max latency (ns)": 991128333, "Mean latency (ns)": 884199803, "Min duration satisfied": "Yes", "Min latency (ns)": 882893269, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1.13, "QPS w/o loadgen overhead": 1.13, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7803, "characteristics.mean": 0.85443, "characteristics.power": 13.06555248618785, "characteristics.power.normalized_per_core": 13.06555248618785, "characteristics.power.normalized_per_processor": 13.06555248618785, "characteristics.samples_per_second": 1.13, "characteristics.samples_per_second.normalized_per_core": 1.13, "characteristics.samples_per_second.normalized_per_processor": 1.13, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9133, "ck_system": "Xavier_NX_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_MaxQ", "system_name": "Auvidea JNX30 Xavier NX (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.12613, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "d4054a8aa7f52e50", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 884117079, "90.00 percentile latency (ns)": 884570991, "90th percentile latency (ns)": 884570991, "95.00 percentile latency (ns)": 884734997, "97.00 percentile latency (ns)": 884882871, "99.00 percentile latency (ns)": 885074323, "99.90 percentile latency (ns)": 885446839, "Max latency (ns)": 991128333, "Mean latency (ns)": 884199803, "Min duration satisfied": "Yes", "Min latency (ns)": 882893269, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1.13, "QPS w/o loadgen overhead": 1.13, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7803, "characteristics.mean": 0.85443, "characteristics.power": 13.06555248618785, "characteristics.power.normalized_per_core": 13.06555248618785, "characteristics.power.normalized_per_processor": 13.06555248618785, "characteristics.samples_per_second": 1.13, "characteristics.samples_per_second.normalized_per_core": 1.13, "characteristics.samples_per_second.normalized_per_processor": 1.13, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9133, "ck_system": "Xavier_NX_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_MaxQ", "system_name": "Auvidea JNX30 Xavier NX (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.12613, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "0cf419bebc864a5a", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 414179613, "90.00 percentile latency (ns)": 417710438, "90th percentile latency (ns)": 417710438, "95.00 percentile latency (ns)": 418741460, "97.00 percentile latency (ns)": 419581017, "99.00 percentile latency (ns)": 420558564, "99.90 percentile latency (ns)": 421895040, "Max latency (ns)": 422402218, "Mean latency (ns)": 414255140, "Min duration satisfied": "Yes", "Min latency (ns)": 406274794, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2.41, "QPS w/o loadgen overhead": 2.41, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.779, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 2.41, "characteristics.samples_per_second.normalized_per_core": 2.41, "characteristics.samples_per_second.normalized_per_processor": 2.41, "characteristics.tumor core": 0.8689, "characteristics.whole tumor": 0.9142, "ck_system": "AGX_Xavier_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_Triton", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2.25225, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "009ed128e5c0301f", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 414179613, "90.00 percentile latency (ns)": 417710438, "90th percentile latency (ns)": 417710438, "95.00 percentile latency (ns)": 418741460, "97.00 percentile latency (ns)": 419581017, "99.00 percentile latency (ns)": 420558564, "99.90 percentile latency (ns)": 421895040, "Max latency (ns)": 422402218, "Mean latency (ns)": 414255140, "Min duration satisfied": "Yes", "Min latency (ns)": 406274794, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2.41, "QPS w/o loadgen overhead": 2.41, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.779, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 2.41, "characteristics.samples_per_second.normalized_per_core": 2.41, "characteristics.samples_per_second.normalized_per_processor": 2.41, "characteristics.tumor core": 0.8689, "characteristics.whole tumor": 0.9142, "ck_system": "AGX_Xavier_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_Triton", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2.25225, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "9bd4b38a5385b518", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 352521204761, "90.00 percentile latency (ns)": 633620566727, "95.00 percentile latency (ns)": 668682539056, "97.00 percentile latency (ns)": 682778637191, "99.00 percentile latency (ns)": 696824736135, "99.90 percentile latency (ns)": 703140212431, "Max latency (ns)": 703835485636, "Mean latency (ns)": 352266702045, "Min duration satisfied": "Yes", "Min latency (ns)": 108457838, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 49.6991, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 49.6991, "characteristics.samples_per_second.normalized_per_core": 49.6991, "characteristics.samples_per_second.normalized_per_processor": 49.6991, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9142, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 34980, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "2a2ed4ae217b7d9a", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 352521204761, "90.00 percentile latency (ns)": 633620566727, "95.00 percentile latency (ns)": 668682539056, "97.00 percentile latency (ns)": 682778637191, "99.00 percentile latency (ns)": 696824736135, "99.90 percentile latency (ns)": 703140212431, "Max latency (ns)": 703835485636, "Mean latency (ns)": 352266702045, "Min duration satisfied": "Yes", "Min latency (ns)": 108457838, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 49.6991, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 49.6991, "characteristics.samples_per_second.normalized_per_core": 49.6991, "characteristics.samples_per_second.normalized_per_processor": 49.6991, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9142, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 34980, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "fe494d881a85001b", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 374252130, "90.00 percentile latency (ns)": 377172739, "90th percentile latency (ns)": 377172739, "95.00 percentile latency (ns)": 378087692, "97.00 percentile latency (ns)": 378696742, "99.00 percentile latency (ns)": 379774422, "99.90 percentile latency (ns)": 380474804, "Max latency (ns)": 380706271, "Mean latency (ns)": 374317483, "Min duration satisfied": "Yes", "Min latency (ns)": 367540314, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2.67, "QPS w/o loadgen overhead": 2.67, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.779, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 2.67, "characteristics.samples_per_second.normalized_per_core": 2.67, "characteristics.samples_per_second.normalized_per_processor": 2.67, "characteristics.tumor core": 0.8689, "characteristics.whole tumor": 0.9142, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2.25225, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "a6f4db8a7ce2d42e", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 374252130, "90.00 percentile latency (ns)": 377172739, "90th percentile latency (ns)": 377172739, "95.00 percentile latency (ns)": 378087692, "97.00 percentile latency (ns)": 378696742, "99.00 percentile latency (ns)": 379774422, "99.90 percentile latency (ns)": 380474804, "Max latency (ns)": 380706271, "Mean latency (ns)": 374317483, "Min duration satisfied": "Yes", "Min latency (ns)": 367540314, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2.67, "QPS w/o loadgen overhead": 2.67, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.779, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 2.67, "characteristics.samples_per_second.normalized_per_core": 2.67, "characteristics.samples_per_second.normalized_per_processor": 2.67, "characteristics.tumor core": 0.8689, "characteristics.whole tumor": 0.9142, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2.25225, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "a4de187d612a4b54", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1621968763374, "90.00 percentile latency (ns)": 2919335675104, "95.00 percentile latency (ns)": 3081531540713, "97.00 percentile latency (ns)": 3146330289621, "99.00 percentile latency (ns)": 3211265031895, "99.90 percentile latency (ns)": 3240428727018, "Max latency (ns)": 3243595439121, "Mean latency (ns)": 1621888902127, "Min duration satisfied": "Yes", "Min latency (ns)": 145355995, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 7.57678, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7802, "characteristics.mean": 0.8544, "characteristics.samples_per_second": 7.57678, "characteristics.samples_per_second.normalized_per_core": 7.57678, "characteristics.samples_per_second.normalized_per_processor": 7.57678, "characteristics.tumor core": 0.8698, "characteristics.whole tumor": 0.9132, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "f596f2284b71a1ad", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1621968763374, "90.00 percentile latency (ns)": 2919335675104, "95.00 percentile latency (ns)": 3081531540713, "97.00 percentile latency (ns)": 3146330289621, "99.00 percentile latency (ns)": 3211265031895, "99.90 percentile latency (ns)": 3240428727018, "Max latency (ns)": 3243595439121, "Mean latency (ns)": 1621888902127, "Min duration satisfied": "Yes", "Min latency (ns)": 145355995, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 7.57678, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7802, "characteristics.mean": 0.8544, "characteristics.samples_per_second": 7.57678, "characteristics.samples_per_second.normalized_per_core": 7.57678, "characteristics.samples_per_second.normalized_per_processor": 7.57678, "characteristics.tumor core": 0.8698, "characteristics.whole tumor": 0.9132, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "f37f3a7d2a45bd8d", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 304674540373, "90.00 percentile latency (ns)": 549299283901, "95.00 percentile latency (ns)": 579861888126, "97.00 percentile latency (ns)": 592101918269, "99.00 percentile latency (ns)": 604340140086, "99.90 percentile latency (ns)": 609829248343, "Max latency (ns)": 610425885891, "Mean latency (ns)": 304748528771, "Min duration satisfied": "Yes", "Min latency (ns)": 58129786, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 57.3043, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 57.3043, "characteristics.samples_per_second.normalized_per_core": 57.3043, "characteristics.samples_per_second.normalized_per_processor": 57.3043, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9142, "ck_system": "A100-PCIe-80GBx1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIe-80GBx1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 34980, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIe-80GBx1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe-80GB, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "341ebf442e6ad9cc", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 304674540373, "90.00 percentile latency (ns)": 549299283901, "95.00 percentile latency (ns)": 579861888126, "97.00 percentile latency (ns)": 592101918269, "99.00 percentile latency (ns)": 604340140086, "99.90 percentile latency (ns)": 609829248343, "Max latency (ns)": 610425885891, "Mean latency (ns)": 304748528771, "Min duration satisfied": "Yes", "Min latency (ns)": 58129786, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 57.3043, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 57.3043, "characteristics.samples_per_second.normalized_per_core": 57.3043, "characteristics.samples_per_second.normalized_per_processor": 57.3043, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9142, "ck_system": "A100-PCIe-80GBx1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIe-80GBx1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 34980, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIe-80GBx1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe-80GB, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "d470583c6aa96e40", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 320989650559, "90.00 percentile latency (ns)": 577925905058, "95.00 percentile latency (ns)": 610044945015, "97.00 percentile latency (ns)": 622892639507, "99.00 percentile latency (ns)": 635736951867, "99.90 percentile latency (ns)": 641508922117, "Max latency (ns)": 642126490343, "Mean latency (ns)": 321012782546, "Min duration satisfied": "Yes", "Min latency (ns)": 37901532, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 61.6701, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 61.6701, "characteristics.samples_per_second.normalized_per_core": 61.6701, "characteristics.samples_per_second.normalized_per_processor": 61.6701, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9142, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 39600, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 60, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "e4fb199de26ae28a", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 320989650559, "90.00 percentile latency (ns)": 577925905058, "95.00 percentile latency (ns)": 610044945015, "97.00 percentile latency (ns)": 622892639507, "99.00 percentile latency (ns)": 635736951867, "99.90 percentile latency (ns)": 641508922117, "Max latency (ns)": 642126490343, "Mean latency (ns)": 321012782546, "Min duration satisfied": "Yes", "Min latency (ns)": 37901532, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 61.6701, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 61.6701, "characteristics.samples_per_second.normalized_per_core": 61.6701, "characteristics.samples_per_second.normalized_per_processor": 61.6701, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9142, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 39600, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 60, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "b104aee889bd7ae0", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 651634855, "90.00 percentile latency (ns)": 652834093, "90th percentile latency (ns)": 652834093, "95.00 percentile latency (ns)": 653337743, "97.00 percentile latency (ns)": 653744527, "99.00 percentile latency (ns)": 675072455, "99.90 percentile latency (ns)": 738290751, "Max latency (ns)": 1090051147, "Mean latency (ns)": 652593626, "Min duration satisfied": "Yes", "Min latency (ns)": 649739087, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1.53, "QPS w/o loadgen overhead": 1.53, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7803, "characteristics.mean": 0.85443, "characteristics.samples_per_second": 1.53, "characteristics.samples_per_second.normalized_per_core": 1.53, "characteristics.samples_per_second.normalized_per_processor": 1.53, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9133, "ck_system": "Xavier_NX_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_Triton", "system_name": "NVIDIA Jetson Xavier NX (TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.12613, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "bbad3f783cb7eb99", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 651634855, "90.00 percentile latency (ns)": 652834093, "90th percentile latency (ns)": 652834093, "95.00 percentile latency (ns)": 653337743, "97.00 percentile latency (ns)": 653744527, "99.00 percentile latency (ns)": 675072455, "99.90 percentile latency (ns)": 738290751, "Max latency (ns)": 1090051147, "Mean latency (ns)": 652593626, "Min duration satisfied": "Yes", "Min latency (ns)": 649739087, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1.53, "QPS w/o loadgen overhead": 1.53, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7803, "characteristics.mean": 0.85443, "characteristics.samples_per_second": 1.53, "characteristics.samples_per_second.normalized_per_core": 1.53, "characteristics.samples_per_second.normalized_per_processor": 1.53, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9133, "ck_system": "Xavier_NX_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_Triton", "system_name": "NVIDIA Jetson Xavier NX (TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.12613, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "fb49a4b9e1f898af", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1634732327101, "90.00 percentile latency (ns)": 2938180946579, "95.00 percentile latency (ns)": 3101020534036, "97.00 percentile latency (ns)": 3166057282502, "99.00 percentile latency (ns)": 3231227048403, "99.90 percentile latency (ns)": 3260498858161, "Max latency (ns)": 3263677399218, "Mean latency (ns)": 1633376121009, "Min duration satisfied": "Yes", "Min latency (ns)": 133147579, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 7.53016, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.6gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7802, "characteristics.mean": 0.8544, "characteristics.samples_per_second": 7.53016, "characteristics.samples_per_second.normalized_per_core": 7.53016, "characteristics.samples_per_second.normalized_per_processor": 7.53016, "characteristics.tumor core": 0.8698, "characteristics.whole tumor": 0.9132, "ck_system": "A30-MIG_1x1g.6gb_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.6gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.6gb_TRT", "system_name": "Gigabyte G482-Z54 (1x A30-MIG-1x1g.6gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7.55, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "20824b23b068bad1", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1634732327101, "90.00 percentile latency (ns)": 2938180946579, "95.00 percentile latency (ns)": 3101020534036, "97.00 percentile latency (ns)": 3166057282502, "99.00 percentile latency (ns)": 3231227048403, "99.90 percentile latency (ns)": 3260498858161, "Max latency (ns)": 3263677399218, "Mean latency (ns)": 1633376121009, "Min duration satisfied": "Yes", "Min latency (ns)": 133147579, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 7.53016, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.6gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7802, "characteristics.mean": 0.8544, "characteristics.samples_per_second": 7.53016, "characteristics.samples_per_second.normalized_per_core": 7.53016, "characteristics.samples_per_second.normalized_per_processor": 7.53016, "characteristics.tumor core": 0.8698, "characteristics.whole tumor": 0.9132, "ck_system": "A30-MIG_1x1g.6gb_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.6gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.6gb_TRT", "system_name": "Gigabyte G482-Z54 (1x A30-MIG-1x1g.6gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7.55, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "f56c34644f07a158", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 470538899, "90.00 percentile latency (ns)": 472475601, "90th percentile latency (ns)": 472475601, "95.00 percentile latency (ns)": 473019246, "97.00 percentile latency (ns)": 473430339, "99.00 percentile latency (ns)": 474604161, "99.90 percentile latency (ns)": 476673596, "Max latency (ns)": 522795481, "Mean latency (ns)": 470769780, "Min duration satisfied": "Yes", "Min latency (ns)": 468405996, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2.12, "QPS w/o loadgen overhead": 2.12, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.779, "characteristics.mean": 0.85403, "characteristics.power": 24.7609534109817, "characteristics.power.normalized_per_core": 24.7609534109817, "characteristics.power.normalized_per_processor": 24.7609534109817, "characteristics.samples_per_second": 2.12, "characteristics.samples_per_second.normalized_per_core": 2.12, "characteristics.samples_per_second.normalized_per_processor": 2.12, "characteristics.tumor core": 0.8689, "characteristics.whole tumor": 0.9142, "ck_system": "AGX_Xavier_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_MaxQ", "system_name": "Auvidea X220-LC AGX Xavier 32GB (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2.25225, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "b159eea0ae67fe8a", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 470538899, "90.00 percentile latency (ns)": 472475601, "90th percentile latency (ns)": 472475601, "95.00 percentile latency (ns)": 473019246, "97.00 percentile latency (ns)": 473430339, "99.00 percentile latency (ns)": 474604161, "99.90 percentile latency (ns)": 476673596, "Max latency (ns)": 522795481, "Mean latency (ns)": 470769780, "Min duration satisfied": "Yes", "Min latency (ns)": 468405996, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2.12, "QPS w/o loadgen overhead": 2.12, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.779, "characteristics.mean": 0.85403, "characteristics.power": 24.7609534109817, "characteristics.power.normalized_per_core": 24.7609534109817, "characteristics.power.normalized_per_processor": 24.7609534109817, "characteristics.samples_per_second": 2.12, "characteristics.samples_per_second.normalized_per_core": 2.12, "characteristics.samples_per_second.normalized_per_processor": 2.12, "characteristics.tumor core": 0.8689, "characteristics.whole tumor": 0.9142, "ck_system": "AGX_Xavier_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_MaxQ", "system_name": "Auvidea X220-LC AGX Xavier 32GB (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2.25225, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "b3bf19fb72489e9f", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 329628351852, "90.00 percentile latency (ns)": 593340965499, "95.00 percentile latency (ns)": 626311917707, "97.00 percentile latency (ns)": 639504674727, "99.00 percentile latency (ns)": 652719200968, "99.90 percentile latency (ns)": 658655352443, "Max latency (ns)": 659296828031, "Mean latency (ns)": 329601193651, "Min duration satisfied": "Yes", "Min latency (ns)": 74769850, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 53.0565, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 53.0565, "characteristics.samples_per_second.normalized_per_core": 53.0565, "characteristics.samples_per_second.normalized_per_processor": 53.0565, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9142, "ck_system": "A100-PCIe-80GBx1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIe-80GBx1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 34980, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIe-80GBx1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe-80GB, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "15ce342528fb03da", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 329628351852, "90.00 percentile latency (ns)": 593340965499, "95.00 percentile latency (ns)": 626311917707, "97.00 percentile latency (ns)": 639504674727, "99.00 percentile latency (ns)": 652719200968, "99.90 percentile latency (ns)": 658655352443, "Max latency (ns)": 659296828031, "Mean latency (ns)": 329601193651, "Min duration satisfied": "Yes", "Min latency (ns)": 74769850, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 53.0565, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 53.0565, "characteristics.samples_per_second.normalized_per_core": 53.0565, "characteristics.samples_per_second.normalized_per_processor": 53.0565, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9142, "ck_system": "A100-PCIe-80GBx1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIe-80GBx1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 34980, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIe-80GBx1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe-80GB, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "ff7a73354269b39b", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 559227958043, "90.00 percentile latency (ns)": 1006790492866, "95.00 percentile latency (ns)": 1062716302202, "97.00 percentile latency (ns)": 1085126179432, "99.00 percentile latency (ns)": 1107532792436, "99.90 percentile latency (ns)": 1117549816261, "Max latency (ns)": 1118641760326, "Mean latency (ns)": 559193759049, "Min duration satisfied": "Yes", "Min latency (ns)": 159757524, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 21.9695, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.854, "characteristics.samples_per_second": 21.9695, "characteristics.samples_per_second.normalized_per_core": 21.9695, "characteristics.samples_per_second.normalized_per_processor": 21.9695, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9141, "ck_system": "A10x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A10x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT_Triton", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 22, "task": "image segmentation", "task2": "image segmentation", "total_cores": 56, "uid": "349f0b2116476ffa", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 559227958043, "90.00 percentile latency (ns)": 1006790492866, "95.00 percentile latency (ns)": 1062716302202, "97.00 percentile latency (ns)": 1085126179432, "99.00 percentile latency (ns)": 1107532792436, "99.90 percentile latency (ns)": 1117549816261, "Max latency (ns)": 1118641760326, "Mean latency (ns)": 559193759049, "Min duration satisfied": "Yes", "Min latency (ns)": 159757524, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 21.9695, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.854, "characteristics.samples_per_second": 21.9695, "characteristics.samples_per_second.normalized_per_core": 21.9695, "characteristics.samples_per_second.normalized_per_processor": 21.9695, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9141, "ck_system": "A10x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A10x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT_Triton", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 22, "task": "image segmentation", "task2": "image segmentation", "total_cores": 56, "uid": "d74d5f97a8da92ce", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 326185734314, "90.00 percentile latency (ns)": 587065056909, "95.00 percentile latency (ns)": 619672614180, "97.00 percentile latency (ns)": 632718527345, "99.00 percentile latency (ns)": 645762090304, "99.90 percentile latency (ns)": 651625046088, "Max latency (ns)": 652251187740, "Mean latency (ns)": 326168335040, "Min duration satisfied": "Yes", "Min latency (ns)": 63823881, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 60.7128, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 60.7128, "characteristics.samples_per_second.normalized_per_core": 60.7128, "characteristics.samples_per_second.normalized_per_processor": 60.7128, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9142, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 39600, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_Triton", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 60, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "e080832c409694f8", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 326185734314, "90.00 percentile latency (ns)": 587065056909, "95.00 percentile latency (ns)": 619672614180, "97.00 percentile latency (ns)": 632718527345, "99.00 percentile latency (ns)": 645762090304, "99.90 percentile latency (ns)": 651625046088, "Max latency (ns)": 652251187740, "Mean latency (ns)": 326168335040, "Min duration satisfied": "Yes", "Min latency (ns)": 63823881, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 60.7128, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85403, "characteristics.samples_per_second": 60.7128, "characteristics.samples_per_second.normalized_per_core": 60.7128, "characteristics.samples_per_second.normalized_per_processor": 60.7128, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9142, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 39600, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_Triton", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 60, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "4fb7abf88e51873b", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 559208965184, "90.00 percentile latency (ns)": 1006847381301, "95.00 percentile latency (ns)": 1062766461460, "97.00 percentile latency (ns)": 1085166728657, "99.00 percentile latency (ns)": 1107572077574, "99.90 percentile latency (ns)": 1117587356914, "Max latency (ns)": 1118679755846, "Mean latency (ns)": 559186313388, "Min duration satisfied": "Yes", "Min latency (ns)": 133888293, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 21.9688, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.854, "characteristics.samples_per_second": 21.9688, "characteristics.samples_per_second.normalized_per_core": 21.9688, "characteristics.samples_per_second.normalized_per_processor": 21.9688, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9141, "ck_system": "A10x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A10x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 22, "task": "image segmentation", "task2": "image segmentation", "total_cores": 56, "uid": "df8ed8bc2ba814a0", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 559208965184, "90.00 percentile latency (ns)": 1006847381301, "95.00 percentile latency (ns)": 1062766461460, "97.00 percentile latency (ns)": 1085166728657, "99.00 percentile latency (ns)": 1107572077574, "99.90 percentile latency (ns)": 1117587356914, "Max latency (ns)": 1118679755846, "Mean latency (ns)": 559186313388, "Min duration satisfied": "Yes", "Min latency (ns)": 133888293, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 21.9688, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.854, "characteristics.samples_per_second": 21.9688, "characteristics.samples_per_second.normalized_per_core": 21.9688, "characteristics.samples_per_second.normalized_per_processor": 21.9688, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9141, "ck_system": "A10x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A10x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 22, "task": "image segmentation", "task2": "image segmentation", "total_cores": 56, "uid": "2dbe38d204b3d906", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 405958997653, "90.00 percentile latency (ns)": 730698312398, "95.00 percentile latency (ns)": 771275043468, "97.00 percentile latency (ns)": 787532241157, "99.00 percentile latency (ns)": 803788296855, "99.90 percentile latency (ns)": 811056292254, "Max latency (ns)": 811848576650, "Mean latency (ns)": 405899527493, "Min duration satisfied": "Yes", "Min latency (ns)": 120013055, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 30.2717, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85407, "characteristics.samples_per_second": 30.2717, "characteristics.samples_per_second.normalized_per_core": 30.2717, "characteristics.samples_per_second.normalized_per_processor": 30.2717, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.9142, "ck_system": "A30x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 30.74, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "053786b62034d14a", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 405958997653, "90.00 percentile latency (ns)": 730698312398, "95.00 percentile latency (ns)": 771275043468, "97.00 percentile latency (ns)": 787532241157, "99.00 percentile latency (ns)": 803788296855, "99.90 percentile latency (ns)": 811056292254, "Max latency (ns)": 811848576650, "Mean latency (ns)": 405899527493, "Min duration satisfied": "Yes", "Min latency (ns)": 120013055, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 30.2717, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85407, "characteristics.samples_per_second": 30.2717, "characteristics.samples_per_second.normalized_per_core": 30.2717, "characteristics.samples_per_second.normalized_per_processor": 30.2717, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.9142, "ck_system": "A30x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 30.74, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "8297462192ae2ac3", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1615526883512, "90.00 percentile latency (ns)": 2907743324120, "95.00 percentile latency (ns)": 3069290322924, "97.00 percentile latency (ns)": 3133834271282, "99.00 percentile latency (ns)": 3198509234151, "99.90 percentile latency (ns)": 3227560991527, "Max latency (ns)": 3230715664978, "Mean latency (ns)": 1615451570013, "Min duration satisfied": "Yes", "Min latency (ns)": 131855852, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 7.60698, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7802, "characteristics.mean": 0.8544, "characteristics.samples_per_second": 7.60698, "characteristics.samples_per_second.normalized_per_core": 7.60698, "characteristics.samples_per_second.normalized_per_processor": 7.60698, "characteristics.tumor core": 0.8698, "characteristics.whole tumor": 0.9132, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "3cdf6e4e9cb2a20c", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1615526883512, "90.00 percentile latency (ns)": 2907743324120, "95.00 percentile latency (ns)": 3069290322924, "97.00 percentile latency (ns)": 3133834271282, "99.00 percentile latency (ns)": 3198509234151, "99.90 percentile latency (ns)": 3227560991527, "Max latency (ns)": 3230715664978, "Mean latency (ns)": 1615451570013, "Min duration satisfied": "Yes", "Min latency (ns)": 131855852, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 7.60698, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7802, "characteristics.mean": 0.8544, "characteristics.samples_per_second": 7.60698, "characteristics.samples_per_second.normalized_per_core": 7.60698, "characteristics.samples_per_second.normalized_per_processor": 7.60698, "characteristics.tumor core": 0.8698, "characteristics.whole tumor": 0.9132, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "48e4b8f56ab293c3", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1605111510449, "90.00 percentile latency (ns)": 2889765199337, "95.00 percentile latency (ns)": 3050290169530, "97.00 percentile latency (ns)": 3114350244337, "99.00 percentile latency (ns)": 3178612283562, "99.90 percentile latency (ns)": 3207470939391, "Max latency (ns)": 3210605905990, "Mean latency (ns)": 1605319690496, "Min duration satisfied": "Yes", "Min latency (ns)": 141783579, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 7.65463, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.6gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7802, "characteristics.mean": 0.8544, "characteristics.samples_per_second": 7.65463, "characteristics.samples_per_second.normalized_per_core": 7.65463, "characteristics.samples_per_second.normalized_per_processor": 7.65463, "characteristics.tumor core": 0.8698, "characteristics.whole tumor": 0.9132, "ck_system": "A30-MIG_1x1g.6gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.6gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.6gb_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30-MIG-1x1g.6gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7.55, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "a7613aec064e1187", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1605111510449, "90.00 percentile latency (ns)": 2889765199337, "95.00 percentile latency (ns)": 3050290169530, "97.00 percentile latency (ns)": 3114350244337, "99.00 percentile latency (ns)": 3178612283562, "99.90 percentile latency (ns)": 3207470939391, "Max latency (ns)": 3210605905990, "Mean latency (ns)": 1605319690496, "Min duration satisfied": "Yes", "Min latency (ns)": 141783579, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 7.65463, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.6gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7802, "characteristics.mean": 0.8544, "characteristics.samples_per_second": 7.65463, "characteristics.samples_per_second.normalized_per_core": 7.65463, "characteristics.samples_per_second.normalized_per_processor": 7.65463, "characteristics.tumor core": 0.8698, "characteristics.whole tumor": 0.9132, "ck_system": "A30-MIG_1x1g.6gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.6gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 24576, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.6gb_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30-MIG-1x1g.6gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7.55, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "fb6149223b38923f", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 335701919825, "90.00 percentile latency (ns)": 605087188441, "95.00 percentile latency (ns)": 638732461644, "97.00 percentile latency (ns)": 652204589753, "99.00 percentile latency (ns)": 665678209262, "99.90 percentile latency (ns)": 671724580232, "Max latency (ns)": 672380065782, "Mean latency (ns)": 335774491528, "Min duration satisfied": "Yes", "Min latency (ns)": 60412769, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 52.0241, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85407, "characteristics.samples_per_second": 52.0241, "characteristics.samples_per_second.normalized_per_core": 52.0241, "characteristics.samples_per_second.normalized_per_processor": 52.0241, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.9142, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 34980, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "a9b127742cb850e4", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 335701919825, "90.00 percentile latency (ns)": 605087188441, "95.00 percentile latency (ns)": 638732461644, "97.00 percentile latency (ns)": 652204589753, "99.00 percentile latency (ns)": 665678209262, "99.90 percentile latency (ns)": 671724580232, "Max latency (ns)": 672380065782, "Mean latency (ns)": 335774491528, "Min duration satisfied": "Yes", "Min latency (ns)": 60412769, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 52.0241, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7789, "characteristics.mean": 0.85407, "characteristics.samples_per_second": 52.0241, "characteristics.samples_per_second.normalized_per_core": 52.0241, "characteristics.samples_per_second.normalized_per_processor": 52.0241, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.9142, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 34980, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "ae96045918feb512", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 627278614, "90.00 percentile latency (ns)": 627724593, "90th percentile latency (ns)": 627724593, "95.00 percentile latency (ns)": 627834657, "97.00 percentile latency (ns)": 627912812, "99.00 percentile latency (ns)": 628251692, "99.90 percentile latency (ns)": 633409519, "Max latency (ns)": 733824848, "Mean latency (ns)": 627407150, "Min duration satisfied": "Yes", "Min latency (ns)": 626261542, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1.59, "QPS w/o loadgen overhead": 1.59, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7803, "characteristics.mean": 0.85443, "characteristics.samples_per_second": 1.59, "characteristics.samples_per_second.normalized_per_core": 1.59, "characteristics.samples_per_second.normalized_per_processor": 1.59, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9133, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.12613, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "fbb576c05c32260e", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 627278614, "90.00 percentile latency (ns)": 627724593, "90th percentile latency (ns)": 627724593, "95.00 percentile latency (ns)": 627834657, "97.00 percentile latency (ns)": 627912812, "99.00 percentile latency (ns)": 628251692, "99.90 percentile latency (ns)": 633409519, "Max latency (ns)": 733824848, "Mean latency (ns)": 627407150, "Min duration satisfied": "Yes", "Min latency (ns)": 626261542, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1.59, "QPS w/o loadgen overhead": 1.59, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7803, "characteristics.mean": 0.85443, "characteristics.samples_per_second": 1.59, "characteristics.samples_per_second.normalized_per_core": 1.59, "characteristics.samples_per_second.normalized_per_processor": 1.59, "characteristics.tumor core": 0.8697, "characteristics.whole tumor": 0.9133, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.12613, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "31bbce238b491aa9", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 415781281395, "90.00 percentile latency (ns)": 748496220365, "95.00 percentile latency (ns)": 790052661731, "97.00 percentile latency (ns)": 806701598634, "99.00 percentile latency (ns)": 823354003286, "99.90 percentile latency (ns)": 830797336406, "Max latency (ns)": 831609784964, "Mean latency (ns)": 415756715011, "Min duration satisfied": "Yes", "Min latency (ns)": 134640513, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 29.5523, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 29.5523, "characteristics.samples_per_second.normalized_per_core": 29.5523, "characteristics.samples_per_second.normalized_per_processor": 29.5523, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "A30x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 30.74, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "67304b2fb1fa172b", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 415781281395, "90.00 percentile latency (ns)": 748496220365, "95.00 percentile latency (ns)": 790052661731, "97.00 percentile latency (ns)": 806701598634, "99.00 percentile latency (ns)": 823354003286, "99.90 percentile latency (ns)": 830797336406, "Max latency (ns)": 831609784964, "Mean latency (ns)": 415756715011, "Min duration satisfied": "Yes", "Min latency (ns)": 134640513, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 29.5523, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 29.5523, "characteristics.samples_per_second.normalized_per_core": 29.5523, "characteristics.samples_per_second.normalized_per_processor": 29.5523, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "A30x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 30.74, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "850cc4478a89a91e", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 818536370, "90.00 percentile latency (ns)": 819081956, "90th percentile latency (ns)": 819081956, "95.00 percentile latency (ns)": 819264972, "97.00 percentile latency (ns)": 819397197, "99.00 percentile latency (ns)": 819628882, "99.90 percentile latency (ns)": 820308174, "Max latency (ns)": 884305309, "Mean latency (ns)": 818618121, "Min duration satisfied": "Yes", "Min latency (ns)": 817293056, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1.22, "QPS w/o loadgen overhead": 1.22, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85377, "characteristics.power": 14.825399761336515, "characteristics.power.normalized_per_core": 14.825399761336515, "characteristics.power.normalized_per_processor": 14.825399761336515, "characteristics.samples_per_second": 1.22, "characteristics.samples_per_second.normalized_per_core": 1.22, "characteristics.samples_per_second.normalized_per_processor": 1.22, "characteristics.tumor core": 0.8683, "characteristics.whole tumor": 0.913, "ck_system": "Xavier_NX_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_MaxQ", "system_name": "NVIDIA Jetson Xavier NX (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.12613, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "ef282d42f1838cb5", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 818536370, "90.00 percentile latency (ns)": 819081956, "90th percentile latency (ns)": 819081956, "95.00 percentile latency (ns)": 819264972, "97.00 percentile latency (ns)": 819397197, "99.00 percentile latency (ns)": 819628882, "99.90 percentile latency (ns)": 820308174, "Max latency (ns)": 884305309, "Mean latency (ns)": 818618121, "Min duration satisfied": "Yes", "Min latency (ns)": 817293056, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1.22, "QPS w/o loadgen overhead": 1.22, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85377, "characteristics.power": 14.825399761336515, "characteristics.power.normalized_per_core": 14.825399761336515, "characteristics.power.normalized_per_processor": 14.825399761336515, "characteristics.samples_per_second": 1.22, "characteristics.samples_per_second.normalized_per_core": 1.22, "characteristics.samples_per_second.normalized_per_processor": 1.22, "characteristics.tumor core": 0.8683, "characteristics.whole tumor": 0.913, "ck_system": "Xavier_NX_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_MaxQ", "system_name": "NVIDIA Jetson Xavier NX (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.12613, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "cfff45ac9ceef83f", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 351749446357, "90.00 percentile latency (ns)": 632421785155, "95.00 percentile latency (ns)": 667497691770, "97.00 percentile latency (ns)": 681545668663, "99.00 percentile latency (ns)": 695587224631, "99.90 percentile latency (ns)": 701886859974, "Max latency (ns)": 702569351516, "Mean latency (ns)": 351484686832, "Min duration satisfied": "Yes", "Min latency (ns)": 111557683, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 49.7887, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 49.7887, "characteristics.samples_per_second.normalized_per_core": 49.7887, "characteristics.samples_per_second.normalized_per_processor": 49.7887, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 34980, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "4a1da0558162bc55", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 350769333406, "90.00 percentile latency (ns)": 631928967140, "95.00 percentile latency (ns)": 666842815382, "97.00 percentile latency (ns)": 680824904434, "99.00 percentile latency (ns)": 694857784923, "99.90 percentile latency (ns)": 701128434258, "Max latency (ns)": 701809068309, "Mean latency (ns)": 351186767095, "Min duration satisfied": "Yes", "Min latency (ns)": 111102196, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 49.8426, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 49.8426, "characteristics.samples_per_second.normalized_per_core": 49.8426, "characteristics.samples_per_second.normalized_per_processor": 49.8426, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 34980, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "2c4deeafcb0cd3c3", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 4697413543332, "90.00 percentile latency (ns)": 8455345747761, "95.00 percentile latency (ns)": 8924990655421, "97.00 percentile latency (ns)": 9112866866421, "99.00 percentile latency (ns)": 9300761158255, "99.90 percentile latency (ns)": 9385352961736, "Max latency (ns)": 9394536825838, "Mean latency (ns)": 4697218384088, "Min duration satisfied": "Yes", "Min latency (ns)": 381047809, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 2.61599, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7796, "characteristics.mean": 0.8542, "characteristics.samples_per_second": 2.61599, "characteristics.samples_per_second.normalized_per_core": 2.61599, "characteristics.samples_per_second.normalized_per_processor": 2.61599, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.914, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 3, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "469bbb7738c624f9", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 4697413543332, "90.00 percentile latency (ns)": 8455345747761, "95.00 percentile latency (ns)": 8924990655421, "97.00 percentile latency (ns)": 9112866866421, "99.00 percentile latency (ns)": 9300761158255, "99.90 percentile latency (ns)": 9385352961736, "Max latency (ns)": 9394536825838, "Mean latency (ns)": 4697218384088, "Min duration satisfied": "Yes", "Min latency (ns)": 381047809, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 2.61599, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7796, "characteristics.mean": 0.8542, "characteristics.samples_per_second": 2.61599, "characteristics.samples_per_second.normalized_per_core": 2.61599, "characteristics.samples_per_second.normalized_per_processor": 2.61599, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.914, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 3, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "c3ef653afcf9410c", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 326604352541, "90.00 percentile latency (ns)": 588082192921, "95.00 percentile latency (ns)": 620754127828, "97.00 percentile latency (ns)": 633818931035, "99.00 percentile latency (ns)": 646889011798, "99.90 percentile latency (ns)": 652761177603, "Max latency (ns)": 653387119937, "Mean latency (ns)": 326646161567, "Min duration satisfied": "Yes", "Min latency (ns)": 43508463, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 60.6073, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 60.6073, "characteristics.samples_per_second.normalized_per_core": 60.6073, "characteristics.samples_per_second.normalized_per_processor": 60.6073, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_edge", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_edge", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 39600, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_edge", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 60, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "20c0f140c2490da5", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 326604352541, "90.00 percentile latency (ns)": 588082192921, "95.00 percentile latency (ns)": 620754127828, "97.00 percentile latency (ns)": 633818931035, "99.00 percentile latency (ns)": 646889011798, "99.90 percentile latency (ns)": 652761177603, "Max latency (ns)": 653387119937, "Mean latency (ns)": 326646161567, "Min duration satisfied": "Yes", "Min latency (ns)": 43508463, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 60.6073, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 60.6073, "characteristics.samples_per_second.normalized_per_core": 60.6073, "characteristics.samples_per_second.normalized_per_processor": 60.6073, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_edge", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_edge", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 39600, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_edge", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 60, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "d879042358b28f88", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1676290554665, "90.00 percentile latency (ns)": 3018373141439, "95.00 percentile latency (ns)": 3186225193810, "97.00 percentile latency (ns)": 3253279289531, "99.00 percentile latency (ns)": 3320501984993, "99.90 percentile latency (ns)": 3350690009897, "Max latency (ns)": 3353968829318, "Mean latency (ns)": 1676619026360, "Min duration satisfied": "Yes", "Min latency (ns)": 143924080, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 7.32744, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85377, "characteristics.samples_per_second": 7.32744, "characteristics.samples_per_second.normalized_per_core": 7.32744, "characteristics.samples_per_second.normalized_per_processor": 7.32744, "characteristics.tumor core": 0.8683, "characteristics.whole tumor": 0.913, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "system_name": "NVIDIA DGX-A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "40947998ff02f617", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1676290554665, "90.00 percentile latency (ns)": 3018373141439, "95.00 percentile latency (ns)": 3186225193810, "97.00 percentile latency (ns)": 3253279289531, "99.00 percentile latency (ns)": 3320501984993, "99.90 percentile latency (ns)": 3350690009897, "Max latency (ns)": 3353968829318, "Mean latency (ns)": 1676619026360, "Min duration satisfied": "Yes", "Min latency (ns)": 143924080, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 7.32744, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85377, "characteristics.samples_per_second": 7.32744, "characteristics.samples_per_second.normalized_per_core": 7.32744, "characteristics.samples_per_second.normalized_per_processor": 7.32744, "characteristics.tumor core": 0.8683, "characteristics.whole tumor": 0.913, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "system_name": "NVIDIA DGX-A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "6926375251bbc9fe", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 713168571, "90.00 percentile latency (ns)": 714337747, "90th percentile latency (ns)": 714337747, "95.00 percentile latency (ns)": 714652117, "97.00 percentile latency (ns)": 714832043, "99.00 percentile latency (ns)": 715203316, "99.90 percentile latency (ns)": 715940003, "Max latency (ns)": 820695210, "Mean latency (ns)": 713333138, "Min duration satisfied": "Yes", "Min latency (ns)": 711439033, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1.4, "QPS w/o loadgen overhead": 1.4, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7796, "characteristics.mean": 0.8542, "characteristics.power": 18.73249247606021, "characteristics.power.normalized_per_core": 18.73249247606021, "characteristics.power.normalized_per_processor": 18.73249247606021, "characteristics.samples_per_second": 1.4, "characteristics.samples_per_second.normalized_per_core": 1.4, "characteristics.samples_per_second.normalized_per_processor": 1.4, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.914, "ck_system": "AGX_Xavier_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_MaxQ", "system_name": "NVIDIA Jetson AGX Xavier 32GB (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2.25225, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "1dbdcb0382435ab9", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 713168571, "90.00 percentile latency (ns)": 714337747, "90th percentile latency (ns)": 714337747, "95.00 percentile latency (ns)": 714652117, "97.00 percentile latency (ns)": 714832043, "99.00 percentile latency (ns)": 715203316, "99.90 percentile latency (ns)": 715940003, "Max latency (ns)": 820695210, "Mean latency (ns)": 713333138, "Min duration satisfied": "Yes", "Min latency (ns)": 711439033, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1.4, "QPS w/o loadgen overhead": 1.4, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7796, "characteristics.mean": 0.8542, "characteristics.power": 18.73249247606021, "characteristics.power.normalized_per_core": 18.73249247606021, "characteristics.power.normalized_per_processor": 18.73249247606021, "characteristics.samples_per_second": 1.4, "characteristics.samples_per_second.normalized_per_core": 1.4, "characteristics.samples_per_second.normalized_per_processor": 1.4, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.914, "ck_system": "AGX_Xavier_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": true, "problem_str": "scenario in meta (singlestream) doesn't match directory (offline)", "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_MaxQ", "system_name": "NVIDIA Jetson AGX Xavier 32GB (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2.25225, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "dab34d6d7c4b118d", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 561243762604, "90.00 percentile latency (ns)": 1010507486873, "95.00 percentile latency (ns)": 1066618410793, "97.00 percentile latency (ns)": 1089099255136, "99.00 percentile latency (ns)": 1111576320438, "99.90 percentile latency (ns)": 1121628754185, "Max latency (ns)": 1122726521892, "Mean latency (ns)": 561232607081, "Min duration satisfied": "Yes", "Min latency (ns)": 164589207, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 21.8896, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.85427, "characteristics.samples_per_second": 21.8896, "characteristics.samples_per_second.normalized_per_core": 21.8896, "characteristics.samples_per_second.normalized_per_processor": 21.8896, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.914, "ck_system": "A10x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A10x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT_Triton", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 22, "task": "image segmentation", "task2": "image segmentation", "total_cores": 56, "uid": "db5cf3b39c3557be", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 561243762604, "90.00 percentile latency (ns)": 1010507486873, "95.00 percentile latency (ns)": 1066618410793, "97.00 percentile latency (ns)": 1089099255136, "99.00 percentile latency (ns)": 1111576320438, "99.90 percentile latency (ns)": 1121628754185, "Max latency (ns)": 1122726521892, "Mean latency (ns)": 561232607081, "Min duration satisfied": "Yes", "Min latency (ns)": 164589207, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 21.8896, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.85427, "characteristics.samples_per_second": 21.8896, "characteristics.samples_per_second.normalized_per_core": 21.8896, "characteristics.samples_per_second.normalized_per_processor": 21.8896, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.914, "ck_system": "A10x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A10x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT_Triton", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 22, "task": "image segmentation", "task2": "image segmentation", "total_cores": 56, "uid": "7d445f996462cf6a", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 333695756126, "90.00 percentile latency (ns)": 600741264394, "95.00 percentile latency (ns)": 634132077328, "97.00 percentile latency (ns)": 647489040362, "99.00 percentile latency (ns)": 660847789029, "99.90 percentile latency (ns)": 666848757028, "Max latency (ns)": 667490109421, "Mean latency (ns)": 333683709980, "Min duration satisfied": "Yes", "Min latency (ns)": 61050496, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 59.3267, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 59.3267, "characteristics.samples_per_second.normalized_per_core": 59.3267, "characteristics.samples_per_second.normalized_per_processor": 59.3267, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 39600, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 60, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "28ec14ea984c7cba", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 333695756126, "90.00 percentile latency (ns)": 600741264394, "95.00 percentile latency (ns)": 634132077328, "97.00 percentile latency (ns)": 647489040362, "99.00 percentile latency (ns)": 660847789029, "99.90 percentile latency (ns)": 666848757028, "Max latency (ns)": 667490109421, "Mean latency (ns)": 333683709980, "Min duration satisfied": "Yes", "Min latency (ns)": 61050496, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 59.3267, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 59.3267, "characteristics.samples_per_second.normalized_per_core": 59.3267, "characteristics.samples_per_second.normalized_per_processor": 59.3267, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 39600, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 60, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "5d2a7416ca578f2a", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 561402339135, "90.00 percentile latency (ns)": 1010927665651, "95.00 percentile latency (ns)": 1067085663542, "97.00 percentile latency (ns)": 1089587749259, "99.00 percentile latency (ns)": 1112085472133, "99.90 percentile latency (ns)": 1122145801031, "Max latency (ns)": 1123241806122, "Mean latency (ns)": 561427660012, "Min duration satisfied": "Yes", "Min latency (ns)": 133033230, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 21.8795, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.85427, "characteristics.samples_per_second": 21.8795, "characteristics.samples_per_second.normalized_per_core": 21.8795, "characteristics.samples_per_second.normalized_per_processor": 21.8795, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.914, "ck_system": "A10x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A10x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 22, "task": "image segmentation", "task2": "image segmentation", "total_cores": 56, "uid": "99cd3c1b12554524", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 561402339135, "90.00 percentile latency (ns)": 1010927665651, "95.00 percentile latency (ns)": 1067085663542, "97.00 percentile latency (ns)": 1089587749259, "99.00 percentile latency (ns)": 1112085472133, "99.90 percentile latency (ns)": 1122145801031, "Max latency (ns)": 1123241806122, "Mean latency (ns)": 561427660012, "Min duration satisfied": "Yes", "Min latency (ns)": 133033230, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 21.8795, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.85427, "characteristics.samples_per_second": 21.8795, "characteristics.samples_per_second.normalized_per_core": 21.8795, "characteristics.samples_per_second.normalized_per_processor": 21.8795, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.914, "ck_system": "A10x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A10x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 22, "task": "image segmentation", "task2": "image segmentation", "total_cores": 56, "uid": "bb8a8a637c4056a5", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 416416063235, "90.00 percentile latency (ns)": 749598551301, "95.00 percentile latency (ns)": 791229254396, "97.00 percentile latency (ns)": 807909881950, "99.00 percentile latency (ns)": 824584071603, "99.90 percentile latency (ns)": 832039690569, "Max latency (ns)": 832853585619, "Mean latency (ns)": 416379147842, "Min duration satisfied": "Yes", "Min latency (ns)": 107886935, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 29.5082, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 29.5082, "characteristics.samples_per_second.normalized_per_core": 29.5082, "characteristics.samples_per_second.normalized_per_processor": 29.5082, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "A30x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 30.74, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "366549879e3863b3", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 416416063235, "90.00 percentile latency (ns)": 749598551301, "95.00 percentile latency (ns)": 791229254396, "97.00 percentile latency (ns)": 807909881950, "99.00 percentile latency (ns)": 824584071603, "99.90 percentile latency (ns)": 832039690569, "Max latency (ns)": 832853585619, "Mean latency (ns)": 416379147842, "Min duration satisfied": "Yes", "Min latency (ns)": 107886935, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 29.5082, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 29.5082, "characteristics.samples_per_second.normalized_per_core": 29.5082, "characteristics.samples_per_second.normalized_per_processor": 29.5082, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "A30x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 30.74, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "89ee945ec8eb1b58", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1673515066778, "90.00 percentile latency (ns)": 3012107206910, "95.00 percentile latency (ns)": 3179418495520, "97.00 percentile latency (ns)": 3246302567509, "99.00 percentile latency (ns)": 3313142408410, "99.90 percentile latency (ns)": 3343188272060, "Max latency (ns)": 3346457690960, "Mean latency (ns)": 1673422122269, "Min duration satisfied": "Yes", "Min latency (ns)": 135914838, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 7.34388, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85377, "characteristics.samples_per_second": 7.34388, "characteristics.samples_per_second.normalized_per_core": 7.34388, "characteristics.samples_per_second.normalized_per_processor": 7.34388, "characteristics.tumor core": 0.8683, "characteristics.whole tumor": 0.913, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "system_name": "NVIDIA DGX-A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "3ee97578b3d94d46", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1673515066778, "90.00 percentile latency (ns)": 3012107206910, "95.00 percentile latency (ns)": 3179418495520, "97.00 percentile latency (ns)": 3246302567509, "99.00 percentile latency (ns)": 3313142408410, "99.90 percentile latency (ns)": 3343188272060, "Max latency (ns)": 3346457690960, "Mean latency (ns)": 1673422122269, "Min duration satisfied": "Yes", "Min latency (ns)": 135914838, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 7.34388, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85377, "characteristics.samples_per_second": 7.34388, "characteristics.samples_per_second.normalized_per_core": 7.34388, "characteristics.samples_per_second.normalized_per_processor": 7.34388, "characteristics.tumor core": 0.8683, "characteristics.whole tumor": 0.913, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "system_name": "NVIDIA DGX-A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 7, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "38ce4777b33ca807", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 338204202168, "90.00 percentile latency (ns)": 609489516518, "95.00 percentile latency (ns)": 643382927596, "97.00 percentile latency (ns)": 656956112397, "99.00 percentile latency (ns)": 670530147442, "99.90 percentile latency (ns)": 676619941542, "Max latency (ns)": 677279160920, "Mean latency (ns)": 338250510201, "Min duration satisfied": "Yes", "Min latency (ns)": 62393060, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 51.6478, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 51.6478, "characteristics.samples_per_second.normalized_per_core": 51.6478, "characteristics.samples_per_second.normalized_per_processor": 51.6478, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 34980, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "4fac8ea0db56b486", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 338204202168, "90.00 percentile latency (ns)": 609489516518, "95.00 percentile latency (ns)": 643382927596, "97.00 percentile latency (ns)": 656956112397, "99.00 percentile latency (ns)": 670530147442, "99.90 percentile latency (ns)": 676619941542, "Max latency (ns)": 677279160920, "Mean latency (ns)": 338250510201, "Min duration satisfied": "Yes", "Min latency (ns)": 62393060, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 51.6478, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.7798, "characteristics.mean": 0.8543, "characteristics.samples_per_second": 51.6478, "characteristics.samples_per_second.normalized_per_core": 51.6478, "characteristics.samples_per_second.normalized_per_processor": 51.6478, "characteristics.tumor core": 0.8691, "characteristics.whole tumor": 0.914, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 34980, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 53, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "3e785b48c3dd45de", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8374294470374, "90.00 percentile latency (ns)": 15073772902283, "95.00 percentile latency (ns)": 15911428873407, "97.00 percentile latency (ns)": 16246060277672, "99.00 percentile latency (ns)": 16581340599971, "99.90 percentile latency (ns)": 16731963269095, "Max latency (ns)": 16748322070072, "Mean latency (ns)": 8374186675498, "Min duration satisfied": "Yes", "Min latency (ns)": 683171756, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 1.46737, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85377, "characteristics.samples_per_second": 1.46737, "characteristics.samples_per_second.normalized_per_core": 1.46737, "characteristics.samples_per_second.normalized_per_processor": 1.46737, "characteristics.tumor core": 0.8683, "characteristics.whole tumor": 0.913, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.5, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "f104a349bf9335ab", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8374294470374, "90.00 percentile latency (ns)": 15073772902283, "95.00 percentile latency (ns)": 15911428873407, "97.00 percentile latency (ns)": 16246060277672, "99.00 percentile latency (ns)": 16581340599971, "99.90 percentile latency (ns)": 16731963269095, "Max latency (ns)": 16748322070072, "Mean latency (ns)": 8374186675498, "Min duration satisfied": "Yes", "Min latency (ns)": 683171756, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 1.46737, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.enhancing tumor": 0.78, "characteristics.mean": 0.85377, "characteristics.samples_per_second": 1.46737, "characteristics.samples_per_second.normalized_per_core": 1.46737, "characteristics.samples_per_second.normalized_per_processor": 1.46737, "characteristics.tumor core": 0.8683, "characteristics.whole tumor": 0.913, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 16, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 24576, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.5, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "d4e561b38f8f4f67", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 293853312005, "90.00 percentile latency (ns)": 529099476565, "95.00 percentile latency (ns)": 558461281503, "97.00 percentile latency (ns)": 570226620836, "99.00 percentile latency (ns)": 581993841588, "99.90 percentile latency (ns)": 587255212074, "Max latency (ns)": 587829129978, "Mean latency (ns)": 293887705515, "Min duration satisfied": "Yes", "Min latency (ns)": 66030712, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 41.8081, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7787, "characteristics.mean": 0.85333, "characteristics.samples_per_second": 41.8081, "characteristics.samples_per_second.normalized_per_core": 41.8081, "characteristics.samples_per_second.normalized_per_processor": 41.8081, "characteristics.tumor core": 0.8693, "characteristics.whole tumor": 0.912, "ck_system": "DGX-A100_A100-SXM4x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1_TRT", "system_name": "NVIDIA DGX-A100 (1x A100-SXM4, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 46, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "daf224dbbeed44e5", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 293853312005, "90.00 percentile latency (ns)": 529099476565, "95.00 percentile latency (ns)": 558461281503, "97.00 percentile latency (ns)": 570226620836, "99.00 percentile latency (ns)": 581993841588, "99.90 percentile latency (ns)": 587255212074, "Max latency (ns)": 587829129978, "Mean latency (ns)": 293887705515, "Min duration satisfied": "Yes", "Min latency (ns)": 66030712, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 41.8081, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7787, "characteristics.mean": 0.85333, "characteristics.samples_per_second": 41.8081, "characteristics.samples_per_second.normalized_per_core": 41.8081, "characteristics.samples_per_second.normalized_per_processor": 41.8081, "characteristics.tumor core": 0.8693, "characteristics.whole tumor": 0.912, "ck_system": "DGX-A100_A100-SXM4x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1_TRT", "system_name": "NVIDIA DGX-A100 (1x A100-SXM4, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 46, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "ccaa91c77034ddb9", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 2368642547665, "90.00 percentile latency (ns)": 4263267531141, "95.00 percentile latency (ns)": 4500136402367, "97.00 percentile latency (ns)": 4594769696462, "99.00 percentile latency (ns)": 4689595771413, "99.90 percentile latency (ns)": 4732188470760, "Max latency (ns)": 4736814178286, "Mean latency (ns)": 2368528884825, "Min duration satisfied": "Yes", "Min latency (ns)": 203249575, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 5.1883, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "5GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4 (1x1g.5gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7793, "characteristics.mean": 0.8536, "characteristics.samples_per_second": 5.1883, "characteristics.samples_per_second.normalized_per_core": 5.1883, "characteristics.samples_per_second.normalized_per_processor": 5.1883, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9125, "ck_system": "DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 8, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "6b166df2d58f942c", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 2368642547665, "90.00 percentile latency (ns)": 4263267531141, "95.00 percentile latency (ns)": 4500136402367, "97.00 percentile latency (ns)": 4594769696462, "99.00 percentile latency (ns)": 4689595771413, "99.90 percentile latency (ns)": 4732188470760, "Max latency (ns)": 4736814178286, "Mean latency (ns)": 2368528884825, "Min duration satisfied": "Yes", "Min latency (ns)": 203249575, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 5.1883, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "5GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4 (1x1g.5gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7793, "characteristics.mean": 0.8536, "characteristics.samples_per_second": 5.1883, "characteristics.samples_per_second.normalized_per_core": 5.1883, "characteristics.samples_per_second.normalized_per_processor": 5.1883, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9125, "ck_system": "DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 8, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "9254db2b70f7d857", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 347035405322, "90.00 percentile latency (ns)": 624854554462, "95.00 percentile latency (ns)": 659565422581, "97.00 percentile latency (ns)": 673474763557, "99.00 percentile latency (ns)": 687378011846, "99.90 percentile latency (ns)": 693599482725, "Max latency (ns)": 694278554107, "Mean latency (ns)": 347046351005, "Min duration satisfied": "Yes", "Min latency (ns)": 134485878, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 35.3979, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7787, "characteristics.mean": 0.85317, "characteristics.samples_per_second": 35.3979, "characteristics.samples_per_second.normalized_per_core": 35.3979, "characteristics.samples_per_second.normalized_per_processor": 35.3979, "characteristics.tumor core": 0.8688, "characteristics.whole tumor": 0.912, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z52 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 41, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "d838e3be3b17ad0f", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 347035405322, "90.00 percentile latency (ns)": 624854554462, "95.00 percentile latency (ns)": 659565422581, "97.00 percentile latency (ns)": 673474763557, "99.00 percentile latency (ns)": 687378011846, "99.90 percentile latency (ns)": 693599482725, "Max latency (ns)": 694278554107, "Mean latency (ns)": 347046351005, "Min duration satisfied": "Yes", "Min latency (ns)": 134485878, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 35.3979, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7787, "characteristics.mean": 0.85317, "characteristics.samples_per_second": 35.3979, "characteristics.samples_per_second.normalized_per_core": 35.3979, "characteristics.samples_per_second.normalized_per_processor": 35.3979, "characteristics.tumor core": 0.8688, "characteristics.whole tumor": 0.912, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z52 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 41, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "c410a9b64504d974", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 5370203625598, "90.00 percentile latency (ns)": 9666248011071, "95.00 percentile latency (ns)": 10203177915558, "97.00 percentile latency (ns)": 10417919707337, "99.00 percentile latency (ns)": 10632936207435, "99.90 percentile latency (ns)": 10729647147583, "Max latency (ns)": 10740102056205, "Mean latency (ns)": 5370188030688, "Min duration satisfied": "Yes", "Min latency (ns)": 437753451, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 2.28825, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7793, "characteristics.mean": 0.85353, "characteristics.samples_per_second": 2.28825, "characteristics.samples_per_second.normalized_per_core": 2.28825, "characteristics.samples_per_second.normalized_per_processor": 2.28825, "characteristics.tumor core": 0.8688, "characteristics.whole tumor": 0.9125, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2", "host_memory_capacity": "32GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2.5, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "8a564bf7bfde3a3b", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 5370203625598, "90.00 percentile latency (ns)": 9666248011071, "95.00 percentile latency (ns)": 10203177915558, "97.00 percentile latency (ns)": 10417919707337, "99.00 percentile latency (ns)": 10632936207435, "99.90 percentile latency (ns)": 10729647147583, "Max latency (ns)": 10740102056205, "Mean latency (ns)": 5370188030688, "Min duration satisfied": "Yes", "Min latency (ns)": 437753451, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 2.28825, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7793, "characteristics.mean": 0.85353, "characteristics.samples_per_second": 2.28825, "characteristics.samples_per_second.normalized_per_core": 2.28825, "characteristics.samples_per_second.normalized_per_processor": 2.28825, "characteristics.tumor core": 0.8688, "characteristics.whole tumor": 0.9125, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2", "host_memory_capacity": "32GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2.5, "task": "image segmentation", "task2": "image segmentation", "total_cores": 8, "uid": "9ffa400fa45d8218", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 325754847818, "90.00 percentile latency (ns)": 586026294969, "95.00 percentile latency (ns)": 618538776049, "97.00 percentile latency (ns)": 631566813077, "99.00 percentile latency (ns)": 644595391692, "99.90 percentile latency (ns)": 650419984979, "Max latency (ns)": 651055643747, "Mean latency (ns)": 325705849149, "Min duration satisfied": "Yes", "Min latency (ns)": 101547780, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 37.7479, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7787, "characteristics.mean": 0.8533, "characteristics.samples_per_second": 37.7479, "characteristics.samples_per_second.normalized_per_core": 37.7479, "characteristics.samples_per_second.normalized_per_processor": 37.7479, "characteristics.tumor core": 0.8692, "characteristics.whole tumor": 0.912, "ck_system": "DGX-A100_A100-SXM4x1_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 46, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "5c1d3a3a0d140c69", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 325754847818, "90.00 percentile latency (ns)": 586026294969, "95.00 percentile latency (ns)": 618538776049, "97.00 percentile latency (ns)": 631566813077, "99.00 percentile latency (ns)": 644595391692, "99.90 percentile latency (ns)": 650419984979, "Max latency (ns)": 651055643747, "Mean latency (ns)": 325705849149, "Min duration satisfied": "Yes", "Min latency (ns)": 101547780, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 37.7479, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7787, "characteristics.mean": 0.8533, "characteristics.samples_per_second": 37.7479, "characteristics.samples_per_second.normalized_per_core": 37.7479, "characteristics.samples_per_second.normalized_per_processor": 37.7479, "characteristics.tumor core": 0.8692, "characteristics.whole tumor": 0.912, "ck_system": "DGX-A100_A100-SXM4x1_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 46, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "9496cc4981b28b76", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 2368819075687, "90.00 percentile latency (ns)": 4263626001490, "95.00 percentile latency (ns)": 4500524874505, "97.00 percentile latency (ns)": 4595169744093, "99.00 percentile latency (ns)": 4690008420227, "99.90 percentile latency (ns)": 4732608348216, "Max latency (ns)": 4737234776295, "Mean latency (ns)": 2368718207859, "Min duration satisfied": "Yes", "Min latency (ns)": 193069929, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 5.18784, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "5GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4 (1x1g.5gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7793, "characteristics.mean": 0.8536, "characteristics.samples_per_second": 5.18784, "characteristics.samples_per_second.normalized_per_core": 5.18784, "characteristics.samples_per_second.normalized_per_processor": 5.18784, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9125, "ck_system": "DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 8, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "81ceec1dbd4f3a89", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 2368819075687, "90.00 percentile latency (ns)": 4263626001490, "95.00 percentile latency (ns)": 4500524874505, "97.00 percentile latency (ns)": 4595169744093, "99.00 percentile latency (ns)": 4690008420227, "99.90 percentile latency (ns)": 4732608348216, "Max latency (ns)": 4737234776295, "Mean latency (ns)": 2368718207859, "Min duration satisfied": "Yes", "Min latency (ns)": 193069929, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 5.18784, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "5GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4 (1x1g.5gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7793, "characteristics.mean": 0.8536, "characteristics.samples_per_second": 5.18784, "characteristics.samples_per_second.normalized_per_core": 5.18784, "characteristics.samples_per_second.normalized_per_processor": 5.18784, "characteristics.tumor core": 0.869, "characteristics.whole tumor": 0.9125, "ck_system": "DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 8, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "d8d44c118f59fbb7", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1743378762771, "90.00 percentile latency (ns)": 3137974667295, "95.00 percentile latency (ns)": 3312194259337, "97.00 percentile latency (ns)": 3381996640782, "99.00 percentile latency (ns)": 3451790741209, "99.90 percentile latency (ns)": 3482991985679, "Max latency (ns)": 3486391101293, "Mean latency (ns)": 1743297722119, "Min duration satisfied": "Yes", "Min latency (ns)": 446590088, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 7.04912, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA T4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7793, "characteristics.mean": 0.85367, "characteristics.samples_per_second": 7.04912, "characteristics.samples_per_second.normalized_per_core": 7.04912, "characteristics.samples_per_second.normalized_per_processor": 7.04912, "characteristics.tumor core": 0.8692, "characteristics.whole tumor": 0.9125, "ck_system": "T4x1_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "ECC off", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/T4x1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/T4x1_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 8, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "e15a79c9edaf4ee5", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1743378762771, "90.00 percentile latency (ns)": 3137974667295, "95.00 percentile latency (ns)": 3312194259337, "97.00 percentile latency (ns)": 3381996640782, "99.00 percentile latency (ns)": 3451790741209, "99.90 percentile latency (ns)": 3482991985679, "Max latency (ns)": 3486391101293, "Mean latency (ns)": 1743297722119, "Min duration satisfied": "Yes", "Min latency (ns)": 446590088, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "Triton_Server", "Samples per second": 7.04912, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA T4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7793, "characteristics.mean": 0.85367, "characteristics.samples_per_second": 7.04912, "characteristics.samples_per_second.normalized_per_core": 7.04912, "characteristics.samples_per_second.normalized_per_processor": 7.04912, "characteristics.tumor core": 0.8692, "characteristics.whole tumor": 0.9125, "ck_system": "T4x1_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "ECC off", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/T4x1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/T4x1_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 8, "task": "image segmentation", "task2": "image segmentation", "total_cores": 240, "uid": "fe3487a3c64414e8", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1705934816126, "90.00 percentile latency (ns)": 3070860651199, "95.00 percentile latency (ns)": 3241369077134, "97.00 percentile latency (ns)": 3309672791208, "99.00 percentile latency (ns)": 3377979462556, "99.90 percentile latency (ns)": 3408525035060, "Max latency (ns)": 3411856580362, "Mean latency (ns)": 1705620331168, "Min duration satisfied": "Yes", "Min latency (ns)": 388924835, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 7.20312, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA T4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7793, "characteristics.mean": 0.85367, "characteristics.samples_per_second": 7.20312, "characteristics.samples_per_second.normalized_per_core": 7.20312, "characteristics.samples_per_second.normalized_per_processor": 7.20312, "characteristics.tumor core": 0.8692, "characteristics.whole tumor": 0.9125, "ck_system": "T4x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "ECC off", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/T4x1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/T4x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x T4, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 8, "task": "image segmentation", "task2": "image segmentation", "total_cores": 56, "uid": "47fa84eeb666baf6", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1705934816126, "90.00 percentile latency (ns)": 3070860651199, "95.00 percentile latency (ns)": 3241369077134, "97.00 percentile latency (ns)": 3309672791208, "99.00 percentile latency (ns)": 3377979462556, "99.90 percentile latency (ns)": 3408525035060, "Max latency (ns)": 3411856580362, "Mean latency (ns)": 1705620331168, "Min duration satisfied": "Yes", "Min latency (ns)": 388924835, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 7.20312, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA T4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7793, "characteristics.mean": 0.85367, "characteristics.samples_per_second": 7.20312, "characteristics.samples_per_second.normalized_per_core": 7.20312, "characteristics.samples_per_second.normalized_per_processor": 7.20312, "characteristics.tumor core": 0.8692, "characteristics.whole tumor": 0.9125, "ck_system": "T4x1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "ECC off", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/T4x1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/T4x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x T4, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 8, "task": "image segmentation", "task2": "image segmentation", "total_cores": 56, "uid": "657cd2024d539f7f", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 320913643283, "90.00 percentile latency (ns)": 578945990056, "95.00 percentile latency (ns)": 611189005808, "97.00 percentile latency (ns)": 624104833456, "99.00 percentile latency (ns)": 637025386397, "99.90 percentile latency (ns)": 642801243635, "Max latency (ns)": 643432610258, "Mean latency (ns)": 321077040363, "Min duration satisfied": "Yes", "Min latency (ns)": 75169645, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 38.1951, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7788, "characteristics.mean": 0.85323, "characteristics.samples_per_second": 38.1951, "characteristics.samples_per_second.normalized_per_core": 38.1951, "characteristics.samples_per_second.normalized_per_processor": 38.1951, "characteristics.tumor core": 0.8689, "characteristics.whole tumor": 0.912, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z52 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 41, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "c684e4cac9f5264e", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 320913643283, "90.00 percentile latency (ns)": 578945990056, "95.00 percentile latency (ns)": 611189005808, "97.00 percentile latency (ns)": 624104833456, "99.00 percentile latency (ns)": 637025386397, "99.90 percentile latency (ns)": 642801243635, "Max latency (ns)": 643432610258, "Mean latency (ns)": 321077040363, "Min duration satisfied": "Yes", "Min latency (ns)": 75169645, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 38.1951, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7788, "characteristics.mean": 0.85323, "characteristics.samples_per_second": 38.1951, "characteristics.samples_per_second.normalized_per_core": 38.1951, "characteristics.samples_per_second.normalized_per_processor": 38.1951, "characteristics.tumor core": 0.8689, "characteristics.whole tumor": 0.912, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "3d-unet-99.9", "input_data_types": "int8", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z52 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 41, "task": "image segmentation", "task2": "image segmentation", "total_cores": 128, "uid": "7b28e414f82c6418", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 9500720587399, "90.00 percentile latency (ns)": 17107406388425, "95.00 percentile latency (ns)": 18057713838424, "97.00 percentile latency (ns)": 18437296644458, "99.00 percentile latency (ns)": 18817679691511, "99.90 percentile latency (ns)": 18988487023959, "Max latency (ns)": 19007046530465, "Mean latency (ns)": 9500292551603, "Min duration satisfied": "Yes", "Min latency (ns)": 763706976, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 1.29299, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7794, "characteristics.mean": 0.85353, "characteristics.samples_per_second": 1.29299, "characteristics.samples_per_second.normalized_per_core": 1.29299, "characteristics.samples_per_second.normalized_per_processor": 1.29299, "characteristics.tumor core": 0.8689, "characteristics.whole tumor": 0.9123, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.0, "formal_model_link": "", "framework": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2", "host_memory_capacity": "8GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.3, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "8446dbeaf655ded5", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 9500720587399, "90.00 percentile latency (ns)": 17107406388425, "95.00 percentile latency (ns)": 18057713838424, "97.00 percentile latency (ns)": 18437296644458, "99.00 percentile latency (ns)": 18817679691511, "99.90 percentile latency (ns)": 18988487023959, "Max latency (ns)": 19007046530465, "Mean latency (ns)": 9500292551603, "Min duration satisfied": "Yes", "Min latency (ns)": 763706976, "Min queries satisfied": "Yes", "Mode": "Performance", "Result is": "VALID", "SUT name": "LWIS_Server", "Samples per second": 1.29299, "Scenario": "offline", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.enhancing tumor": 0.7794, "characteristics.mean": 0.85353, "characteristics.samples_per_second": 1.29299, "characteristics.samples_per_second.normalized_per_core": 1.29299, "characteristics.samples_per_second.normalized_per_processor": 1.29299, "characteristics.tumor core": 0.8689, "characteristics.whole tumor": 0.9123, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "BraTS 2019", "dataset_link": "https://www.med.upenn.edu/cbica/brats2019/data.html", "dim_x_default": "characteristics.samples_per_second", "dim_x_maximize": true, "dim_y_default": "characteristics.mean", "dim_y_maximize": true, "division": "closed", "formal_model": "3d-unet", "formal_model_accuracy": 99.9, "formal_model_link": "", "framework": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2", "host_memory_capacity": "8GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "3d-unet-99.9", "input_data_types": "fp16", "key.accuracy": "characteristics.mean", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 16, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 24576, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "224_224_160_dyanmic_bs.onnx", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1.3, "task": "image segmentation", "task2": "image segmentation", "total_cores": 6, "uid": "174f25a2fdb4a69c", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" } ]