[ { "50.00 percentile latency (ns)": 271640, "90.00 percentile latency (ns)": 278560, "90th percentile latency (ns)": 278560, "95.00 percentile latency (ns)": 281070, "97.00 percentile latency (ns)": 283581, "99.00 percentile latency (ns)": 294460, "99.90 percentile latency (ns)": 322620, "Max latency (ns)": 12400815, "Mean latency (ns)": 273676, "Min duration satisfied": "Yes", "Min latency (ns)": 250830, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3564.09, "QPS w/o loadgen overhead": 3653.96, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.27856, "characteristics.90th_percentile_latency_ns": 278560.0, "characteristics.90th_percentile_latency_s": 0.00027856, "characteristics.90th_percentile_latency_us": 278.56, "characteristics.mAP": 22.914, "ck_system": "A30x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2941.18, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "411a663698e4c058", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 2813567, "90.00 percentile latency (ns)": 2865024, "90th percentile latency (ns)": 2865024, "95.00 percentile latency (ns)": 2879938, "97.00 percentile latency (ns)": 2888579, "99.00 percentile latency (ns)": 2905174, "99.90 percentile latency (ns)": 3235736, "Max latency (ns)": 11979010, "Mean latency (ns)": 2815114, "Min duration satisfied": "Yes", "Min latency (ns)": 2704157, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 351.83, "QPS w/o loadgen overhead": 355.23, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 2.865024, "characteristics.90th_percentile_latency_ns": 2865024.0, "characteristics.90th_percentile_latency_s": 0.002865024, "characteristics.90th_percentile_latency_us": 2865.024, "characteristics.mAP": 20.118, "ck_system": "A30x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 326.413, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "2486ee25e5d44a08", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1700174, "90.00 percentile latency (ns)": 1715565, "90th percentile latency (ns)": 1715565, "95.00 percentile latency (ns)": 1721901, "97.00 percentile latency (ns)": 1727340, "99.00 percentile latency (ns)": 1743661, "99.90 percentile latency (ns)": 6109611, "Max latency (ns)": 13372062, "Mean latency (ns)": 1710672, "Min duration satisfied": "Yes", "Min latency (ns)": 1629900, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 579.04, "QPS w/o loadgen overhead": 584.57, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.715565, "characteristics.90th_percentile_latency_ns": 1715565.0, "characteristics.90th_percentile_latency_s": 0.001715565, "characteristics.90th_percentile_latency_us": 1715.565, "characteristics.mAP": 22.914, "characteristics.power": 0.017184518433491714, "characteristics.power.normalized_per_core": 0.017184518433491714, "characteristics.power.normalized_per_processor": 0.017184518433491714, "ck_system": "Xavier_NX_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_MaxQ", "system_name": "Auvidea JNX30 Xavier NX (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 500, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "e14f4db5fe37c3b4", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 53648710, "90.00 percentile latency (ns)": 53791398, "90th percentile latency (ns)": 53791398, "95.00 percentile latency (ns)": 53978528, "97.00 percentile latency (ns)": 54178213, "99.00 percentile latency (ns)": 54440421, "99.90 percentile latency (ns)": 55017317, "Max latency (ns)": 61566582, "Mean latency (ns)": 53680318, "Min duration satisfied": "Yes", "Min latency (ns)": 53352826, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 18.62, "QPS w/o loadgen overhead": 18.63, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 53.791398, "characteristics.90th_percentile_latency_ns": 53791398.0, "characteristics.90th_percentile_latency_s": 0.053791398, "characteristics.90th_percentile_latency_us": 53791.398, "characteristics.mAP": 20.113, "characteristics.power": 0.7449962947449821, "characteristics.power.normalized_per_core": 0.7449962947449821, "characteristics.power.normalized_per_processor": 0.7449962947449821, "ck_system": "Xavier_NX_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_MaxQ", "system_name": "Auvidea JNX30 Xavier NX (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 16.9618, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "fb9faffa54786a6e", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1189655, "90.00 percentile latency (ns)": 1277755, "90th percentile latency (ns)": 1277755, "95.00 percentile latency (ns)": 1299129, "97.00 percentile latency (ns)": 1314653, "99.00 percentile latency (ns)": 1351357, "99.90 percentile latency (ns)": 1460803, "Max latency (ns)": 23035513, "Mean latency (ns)": 1202114, "Min duration satisfied": "Yes", "Min latency (ns)": 1082705, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 822.7, "QPS w/o loadgen overhead": 831.87, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.277755, "characteristics.90th_percentile_latency_ns": 1277755.0, "characteristics.90th_percentile_latency_s": 0.001277755, "characteristics.90th_percentile_latency_us": 1277.755, "characteristics.mAP": 22.914, "ck_system": "AGX_Xavier_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_Triton", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 666.667, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "c2271fc470b9b382", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 26350429, "90.00 percentile latency (ns)": 26514993, "90th percentile latency (ns)": 26514993, "95.00 percentile latency (ns)": 26599190, "97.00 percentile latency (ns)": 26903819, "99.00 percentile latency (ns)": 27132885, "99.90 percentile latency (ns)": 27421242, "Max latency (ns)": 28079525, "Mean latency (ns)": 26381120, "Min duration satisfied": "Yes", "Min latency (ns)": 26100037, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 37.88, "QPS w/o loadgen overhead": 37.91, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 26.514993, "characteristics.90th_percentile_latency_ns": 26514993.0, "characteristics.90th_percentile_latency_s": 0.026514993, "characteristics.90th_percentile_latency_us": 26514.993, "characteristics.mAP": 20.113, "ck_system": "AGX_Xavier_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_Triton", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 33.9236, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "fffe5f7f4553be68", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 269858, "90.00 percentile latency (ns)": 275229, "90th percentile latency (ns)": 275229, "95.00 percentile latency (ns)": 277849, "97.00 percentile latency (ns)": 280729, "99.00 percentile latency (ns)": 294178, "99.90 percentile latency (ns)": 313459, "Max latency (ns)": 9814999, "Mean latency (ns)": 270992, "Min duration satisfied": "Yes", "Min latency (ns)": 248829, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3598.69, "QPS w/o loadgen overhead": 3690.15, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.275229, "characteristics.90th_percentile_latency_ns": 275229.0, "characteristics.90th_percentile_latency_s": 0.000275229, "characteristics.90th_percentile_latency_us": 275.229, "characteristics.mAP": 22.914, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "73df44dbac0c5448", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1863631, "90.00 percentile latency (ns)": 1889780, "90th percentile latency (ns)": 1889780, "95.00 percentile latency (ns)": 1897711, "97.00 percentile latency (ns)": 1902710, "99.00 percentile latency (ns)": 1913750, "99.90 percentile latency (ns)": 2460257, "Max latency (ns)": 23111499, "Mean latency (ns)": 1866695, "Min duration satisfied": "Yes", "Min latency (ns)": 1802680, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 525.73, "QPS w/o loadgen overhead": 535.71, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.88978, "characteristics.90th_percentile_latency_ns": 1889780.0, "characteristics.90th_percentile_latency_s": 0.00188978, "characteristics.90th_percentile_latency_us": 1889.78, "characteristics.mAP": 20.118, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "411ca754d826fa27", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 979151, "90.00 percentile latency (ns)": 1002352, "90th percentile latency (ns)": 1002352, "95.00 percentile latency (ns)": 1046258, "97.00 percentile latency (ns)": 1114645, "99.00 percentile latency (ns)": 1303613, "99.90 percentile latency (ns)": 1707665, "Max latency (ns)": 5011531, "Mean latency (ns)": 991244, "Min duration satisfied": "Yes", "Min latency (ns)": 940011, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 996.59, "QPS w/o loadgen overhead": 1008.83, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.002352, "characteristics.90th_percentile_latency_ns": 1002352.0, "characteristics.90th_percentile_latency_s": 0.001002352, "characteristics.90th_percentile_latency_us": 1002.352, "characteristics.mAP": 22.914, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 666.667, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "c509e1843c1ad371", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 25791133, "90.00 percentile latency (ns)": 26134242, "90th percentile latency (ns)": 26134242, "95.00 percentile latency (ns)": 26285410, "97.00 percentile latency (ns)": 26391608, "99.00 percentile latency (ns)": 26566127, "99.90 percentile latency (ns)": 26807760, "Max latency (ns)": 27245381, "Mean latency (ns)": 25850855, "Min duration satisfied": "Yes", "Min latency (ns)": 25577152, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 38.66, "QPS w/o loadgen overhead": 38.68, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 26.134242, "characteristics.90th_percentile_latency_ns": 26134242.0, "characteristics.90th_percentile_latency_s": 0.026134242, "characteristics.90th_percentile_latency_us": 26134.242, "characteristics.mAP": 20.113, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 33.9236, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "baeef791423c9221", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 390389, "90.00 percentile latency (ns)": 394617, "90th percentile latency (ns)": 394617, "95.00 percentile latency (ns)": 396059, "97.00 percentile latency (ns)": 397192, "99.00 percentile latency (ns)": 399816, "99.90 percentile latency (ns)": 455311, "Max latency (ns)": 5917472, "Mean latency (ns)": 390782, "Min duration satisfied": "Yes", "Min latency (ns)": 370090, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2523.8, "QPS w/o loadgen overhead": 2558.97, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.394617, "characteristics.90th_percentile_latency_ns": 394617.0, "characteristics.90th_percentile_latency_s": 0.000394617, "characteristics.90th_percentile_latency_us": 394.617, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "dfc414a28b13a9ee", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8088539, "90.00 percentile latency (ns)": 8113056, "90th percentile latency (ns)": 8113056, "95.00 percentile latency (ns)": 8150506, "97.00 percentile latency (ns)": 8167558, "99.00 percentile latency (ns)": 8188208, "99.90 percentile latency (ns)": 9358571, "Max latency (ns)": 24331752, "Mean latency (ns)": 8095736, "Min duration satisfied": "Yes", "Min latency (ns)": 8051308, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 123.42, "QPS w/o loadgen overhead": 123.52, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 8.113056, "characteristics.90th_percentile_latency_ns": 8113056.0, "characteristics.90th_percentile_latency_s": 0.008113056, "characteristics.90th_percentile_latency_us": 8113.056, "characteristics.mAP": 20.118, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "ac794765a34227e6", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 301817, "90.00 percentile latency (ns)": 310557, "90th percentile latency (ns)": 310557, "95.00 percentile latency (ns)": 314398, "97.00 percentile latency (ns)": 319427, "99.00 percentile latency (ns)": 334748, "99.90 percentile latency (ns)": 358907, "Max latency (ns)": 14438705, "Mean latency (ns)": 303398, "Min duration satisfied": "Yes", "Min latency (ns)": 261288, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3203.8, "QPS w/o loadgen overhead": 3296.0, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.310557, "characteristics.90th_percentile_latency_ns": 310557.0, "characteristics.90th_percentile_latency_s": 0.000310557, "characteristics.90th_percentile_latency_us": 310.557, "characteristics.mAP": 22.914, "ck_system": "A100-PCIe-80GBx1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIe-80GBx1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIe-80GBx1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe-80GB, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "153103b5e04c668e", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1924313, "90.00 percentile latency (ns)": 1960403, "90th percentile latency (ns)": 1960403, "95.00 percentile latency (ns)": 1968854, "97.00 percentile latency (ns)": 1973404, "99.00 percentile latency (ns)": 1981193, "99.90 percentile latency (ns)": 2368127, "Max latency (ns)": 10309464, "Mean latency (ns)": 1927292, "Min duration satisfied": "Yes", "Min latency (ns)": 1831525, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 509.74, "QPS w/o loadgen overhead": 518.86, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.960403, "characteristics.90th_percentile_latency_ns": 1960403.0, "characteristics.90th_percentile_latency_s": 0.001960403, "characteristics.90th_percentile_latency_us": 1960.403, "characteristics.mAP": 20.118, "ck_system": "A100-PCIe-80GBx1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIe-80GBx1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIe-80GBx1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe-80GB, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "6903ecb44d6bfc71", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 286607, "90.00 percentile latency (ns)": 291266, "90th percentile latency (ns)": 291266, "95.00 percentile latency (ns)": 294282, "97.00 percentile latency (ns)": 296567, "99.00 percentile latency (ns)": 299412, "99.90 percentile latency (ns)": 319890, "Max latency (ns)": 5919170, "Mean latency (ns)": 287942, "Min duration satisfied": "Yes", "Min latency (ns)": 252143, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3407.47, "QPS w/o loadgen overhead": 3472.92, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.291266, "characteristics.90th_percentile_latency_ns": 291266.0, "characteristics.90th_percentile_latency_s": 0.000291266, "characteristics.90th_percentile_latency_us": 291.266, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "55ee435ba17727ba", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1696020, "90.00 percentile latency (ns)": 1715016, "90th percentile latency (ns)": 1715016, "95.00 percentile latency (ns)": 1719084, "97.00 percentile latency (ns)": 1721448, "99.00 percentile latency (ns)": 1727650, "99.90 percentile latency (ns)": 2746821, "Max latency (ns)": 7304858, "Mean latency (ns)": 1699410, "Min duration satisfied": "Yes", "Min latency (ns)": 1649633, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 586.93, "QPS w/o loadgen overhead": 588.44, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.715016, "characteristics.90th_percentile_latency_ns": 1715016.0, "characteristics.90th_percentile_latency_s": 0.001715016, "characteristics.90th_percentile_latency_us": 1715.016, "characteristics.mAP": 20.118, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "33bc434eb926b596", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1570214, "90.00 percentile latency (ns)": 1618599, "90th percentile latency (ns)": 1618599, "95.00 percentile latency (ns)": 1638152, "97.00 percentile latency (ns)": 1657641, "99.00 percentile latency (ns)": 1758571, "99.90 percentile latency (ns)": 6371739, "Max latency (ns)": 38654183, "Mean latency (ns)": 1587329, "Min duration satisfied": "Yes", "Min latency (ns)": 1410274, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 622.55, "QPS w/o loadgen overhead": 629.99, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.618599, "characteristics.90th_percentile_latency_ns": 1618599.0, "characteristics.90th_percentile_latency_s": 0.001618599, "characteristics.90th_percentile_latency_us": 1618.599, "characteristics.mAP": 22.914, "ck_system": "Xavier_NX_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_Triton", "system_name": "NVIDIA Jetson Xavier NX (TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 500, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "1a26b24e40a2b12d", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 42752236, "90.00 percentile latency (ns)": 43528253, "90th percentile latency (ns)": 43528253, "95.00 percentile latency (ns)": 43709249, "97.00 percentile latency (ns)": 43809699, "99.00 percentile latency (ns)": 44009896, "99.90 percentile latency (ns)": 47747702, "Max latency (ns)": 61651038, "Mean latency (ns)": 42922477, "Min duration satisfied": "Yes", "Min latency (ns)": 42520648, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 23.29, "QPS w/o loadgen overhead": 23.3, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 43.528253, "characteristics.90th_percentile_latency_ns": 43528253.0, "characteristics.90th_percentile_latency_s": 0.043528253, "characteristics.90th_percentile_latency_us": 43528.253, "characteristics.mAP": 20.113, "ck_system": "Xavier_NX_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_Triton", "system_name": "NVIDIA Jetson Xavier NX (TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 16.9618, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "fba9fa2f70842918", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 439990, "90.00 percentile latency (ns)": 449410, "90th percentile latency (ns)": 449410, "95.00 percentile latency (ns)": 464000, "97.00 percentile latency (ns)": 468770, "99.00 percentile latency (ns)": 474980, "99.90 percentile latency (ns)": 562851, "Max latency (ns)": 9439527, "Mean latency (ns)": 441833, "Min duration satisfied": "Yes", "Min latency (ns)": 392600, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2160.68, "QPS w/o loadgen overhead": 2263.3, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.6gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.44941, "characteristics.90th_percentile_latency_ns": 449410.0, "characteristics.90th_percentile_latency_s": 0.00044941, "characteristics.90th_percentile_latency_us": 449.41, "characteristics.mAP": 22.914, "ck_system": "A30-MIG_1x1g.6gb_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.6gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.6gb_TRT", "system_name": "Gigabyte G482-Z54 (1x A30-MIG-1x1g.6gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2220.84, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "17829b5b40d095c7", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8376701, "90.00 percentile latency (ns)": 8401831, "90th percentile latency (ns)": 8401831, "95.00 percentile latency (ns)": 8408591, "97.00 percentile latency (ns)": 8413101, "99.00 percentile latency (ns)": 8421781, "99.90 percentile latency (ns)": 8678240, "Max latency (ns)": 13162131, "Mean latency (ns)": 8372906, "Min duration satisfied": "Yes", "Min latency (ns)": 8253480, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 118.86, "QPS w/o loadgen overhead": 119.43, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.6gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 8.401831, "characteristics.90th_percentile_latency_ns": 8401831.0, "characteristics.90th_percentile_latency_s": 0.008401831, "characteristics.90th_percentile_latency_us": 8401.831, "characteristics.mAP": 20.118, "ck_system": "A30-MIG_1x1g.6gb_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.6gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.6gb_TRT", "system_name": "Gigabyte G482-Z54 (1x A30-MIG-1x1g.6gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 116.033, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "7f3a495d42546ccb", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1191581, "90.00 percentile latency (ns)": 1216959, "90th percentile latency (ns)": 1216959, "95.00 percentile latency (ns)": 1229856, "97.00 percentile latency (ns)": 1243680, "99.00 percentile latency (ns)": 1511182, "99.90 percentile latency (ns)": 2154320, "Max latency (ns)": 13573566, "Mean latency (ns)": 1203004, "Min duration satisfied": "Yes", "Min latency (ns)": 1148699, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 823.55, "QPS w/o loadgen overhead": 831.25, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.216959, "characteristics.90th_percentile_latency_ns": 1216959.0, "characteristics.90th_percentile_latency_s": 0.001216959, "characteristics.90th_percentile_latency_us": 1216.959, "characteristics.mAP": 22.914, "characteristics.power": 0.021986879970857912, "characteristics.power.normalized_per_core": 0.021986879970857912, "characteristics.power.normalized_per_processor": 0.021986879970857912, "ck_system": "AGX_Xavier_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_MaxQ", "system_name": "Auvidea X220-LC AGX Xavier 32GB (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 666.667, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "58bc19d2526eccdb", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 33683905, "90.00 percentile latency (ns)": 33894763, "90th percentile latency (ns)": 33894763, "95.00 percentile latency (ns)": 34168249, "97.00 percentile latency (ns)": 34341922, "99.00 percentile latency (ns)": 34534860, "99.90 percentile latency (ns)": 34741623, "Max latency (ns)": 42903665, "Mean latency (ns)": 33736203, "Min duration satisfied": "Yes", "Min latency (ns)": 33521689, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 29.63, "QPS w/o loadgen overhead": 29.64, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 33.894763, "characteristics.90th_percentile_latency_ns": 33894763.0, "characteristics.90th_percentile_latency_s": 0.033894763, "characteristics.90th_percentile_latency_us": 33894.763, "characteristics.mAP": 20.113, "characteristics.power": 0.7905666809554831, "characteristics.power.normalized_per_core": 0.7905666809554831, "characteristics.power.normalized_per_processor": 0.7905666809554831, "ck_system": "AGX_Xavier_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_MaxQ", "system_name": "Auvidea X220-LC AGX Xavier 32GB (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 33.9236, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "4bd73e5e913d4038", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 267048, "90.00 percentile latency (ns)": 272458, "90th percentile latency (ns)": 272458, "95.00 percentile latency (ns)": 274627, "97.00 percentile latency (ns)": 276518, "99.00 percentile latency (ns)": 283367, "99.90 percentile latency (ns)": 301919, "Max latency (ns)": 8278370, "Mean latency (ns)": 267899, "Min duration satisfied": "Yes", "Min latency (ns)": 245648, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3620.16, "QPS w/o loadgen overhead": 3732.75, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.272458, "characteristics.90th_percentile_latency_ns": 272458.0, "characteristics.90th_percentile_latency_s": 0.000272458, "characteristics.90th_percentile_latency_us": 272.458, "characteristics.mAP": 22.914, "ck_system": "A100-PCIe-80GBx1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIe-80GBx1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIe-80GBx1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe-80GB, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "aef97bb41eea8f95", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1875945, "90.00 percentile latency (ns)": 1917814, "90th percentile latency (ns)": 1917814, "95.00 percentile latency (ns)": 2024273, "97.00 percentile latency (ns)": 2050804, "99.00 percentile latency (ns)": 2106133, "99.90 percentile latency (ns)": 2319381, "Max latency (ns)": 39110736, "Mean latency (ns)": 1893286, "Min duration satisfied": "Yes", "Min latency (ns)": 1807986, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 518.21, "QPS w/o loadgen overhead": 528.18, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.917814, "characteristics.90th_percentile_latency_ns": 1917814.0, "characteristics.90th_percentile_latency_s": 0.001917814, "characteristics.90th_percentile_latency_us": 1917.814, "characteristics.mAP": 20.118, "ck_system": "A100-PCIe-80GBx1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIe-80GBx1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIe-80GBx1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe-80GB, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "34ac3ff2a479d9cb", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 292919, "90.00 percentile latency (ns)": 302983, "90th percentile latency (ns)": 302983, "95.00 percentile latency (ns)": 305429, "97.00 percentile latency (ns)": 306944, "99.00 percentile latency (ns)": 310325, "99.90 percentile latency (ns)": 321304, "Max latency (ns)": 22797097, "Mean latency (ns)": 295016, "Min duration satisfied": "Yes", "Min latency (ns)": 273788, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3298.5, "QPS w/o loadgen overhead": 3389.65, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.302983, "characteristics.90th_percentile_latency_ns": 302983.0, "characteristics.90th_percentile_latency_s": 0.000302983, "characteristics.90th_percentile_latency_us": 302.983, "characteristics.mAP": 22.914, "ck_system": "A10x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A10x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT_Triton", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2680.97, "task": "object detection", "task2": "object detection", "total_cores": 56, "uid": "ac964aaba83ac7d3", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 3874354, "90.00 percentile latency (ns)": 3914304, "90th percentile latency (ns)": 3914304, "95.00 percentile latency (ns)": 3927061, "97.00 percentile latency (ns)": 3935967, "99.00 percentile latency (ns)": 3954906, "99.90 percentile latency (ns)": 5143088, "Max latency (ns)": 26722145, "Mean latency (ns)": 3879103, "Min duration satisfied": "Yes", "Min latency (ns)": 3345010, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 252.63, "QPS w/o loadgen overhead": 257.79, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 3.914304, "characteristics.90th_percentile_latency_ns": 3914304.0, "characteristics.90th_percentile_latency_s": 0.003914304, "characteristics.90th_percentile_latency_us": 3914.304, "characteristics.mAP": 20.113, "ck_system": "A10x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A10x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT_Triton", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 228.833, "task": "object detection", "task2": "object detection", "total_cores": 56, "uid": "2dd5947978ff4611", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 240892, "90.00 percentile latency (ns)": 248296, "90th percentile latency (ns)": 248296, "95.00 percentile latency (ns)": 251011, "97.00 percentile latency (ns)": 253265, "99.00 percentile latency (ns)": 257323, "99.90 percentile latency (ns)": 267502, "Max latency (ns)": 17807893, "Mean latency (ns)": 242685, "Min duration satisfied": "Yes", "Min latency (ns)": 228077, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 4065.78, "QPS w/o loadgen overhead": 4120.57, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.248296, "characteristics.90th_percentile_latency_ns": 248296.0, "characteristics.90th_percentile_latency_s": 0.000248296, "characteristics.90th_percentile_latency_us": 248.296, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_Triton", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "92649428b0222eb8", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1669251, "90.00 percentile latency (ns)": 1689348, "90th percentile latency (ns)": 1689348, "95.00 percentile latency (ns)": 1694538, "97.00 percentile latency (ns)": 1699186, "99.00 percentile latency (ns)": 1721027, "99.90 percentile latency (ns)": 1797070, "Max latency (ns)": 19190536, "Mean latency (ns)": 1674582, "Min duration satisfied": "Yes", "Min latency (ns)": 1633734, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 595.64, "QPS w/o loadgen overhead": 597.16, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.689348, "characteristics.90th_percentile_latency_ns": 1689348.0, "characteristics.90th_percentile_latency_s": 0.001689348, "characteristics.90th_percentile_latency_us": 1689.348, "characteristics.mAP": 20.118, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_Triton", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "33c4cb7a59badf9b", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 322236, "90.00 percentile latency (ns)": 334000, "90th percentile latency (ns)": 334000, "95.00 percentile latency (ns)": 336445, "97.00 percentile latency (ns)": 338448, "99.00 percentile latency (ns)": 343430, "99.90 percentile latency (ns)": 362298, "Max latency (ns)": 4893572, "Mean latency (ns)": 323255, "Min duration satisfied": "Yes", "Min latency (ns)": 291211, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3031.21, "QPS w/o loadgen overhead": 3093.53, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.334, "characteristics.90th_percentile_latency_ns": 334000.0, "characteristics.90th_percentile_latency_s": 0.000334, "characteristics.90th_percentile_latency_us": 334.0, "characteristics.mAP": 22.914, "ck_system": "A10x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A10x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2680.97, "task": "object detection", "task2": "object detection", "total_cores": 56, "uid": "24af41eb35464b8f", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 4028272, "90.00 percentile latency (ns)": 4072390, "90th percentile latency (ns)": 4072390, "95.00 percentile latency (ns)": 4085273, "97.00 percentile latency (ns)": 4093885, "99.00 percentile latency (ns)": 4111648, "99.90 percentile latency (ns)": 4170902, "Max latency (ns)": 9394994, "Mean latency (ns)": 4023929, "Min duration satisfied": "Yes", "Min latency (ns)": 3549031, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 243.9, "QPS w/o loadgen overhead": 248.51, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 4.07239, "characteristics.90th_percentile_latency_ns": 4072390.0, "characteristics.90th_percentile_latency_s": 0.00407239, "characteristics.90th_percentile_latency_us": 4072.39, "characteristics.mAP": 20.113, "ck_system": "A10x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A10x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 228.833, "task": "object detection", "task2": "object detection", "total_cores": 56, "uid": "41f3965b53495192", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 310131, "90.00 percentile latency (ns)": 320249, "90th percentile latency (ns)": 320249, "95.00 percentile latency (ns)": 333980, "97.00 percentile latency (ns)": 341100, "99.00 percentile latency (ns)": 348101, "99.90 percentile latency (ns)": 489419, "Max latency (ns)": 6084024, "Mean latency (ns)": 312703, "Min duration satisfied": "Yes", "Min latency (ns)": 262020, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3108.34, "QPS w/o loadgen overhead": 3197.92, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.320249, "characteristics.90th_percentile_latency_ns": 320249.0, "characteristics.90th_percentile_latency_s": 0.000320249, "characteristics.90th_percentile_latency_us": 320.249, "characteristics.mAP": 22.914, "ck_system": "A30x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2941.18, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "aca11729c53f512c", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 2907938, "90.00 percentile latency (ns)": 2945716, "90th percentile latency (ns)": 2945716, "95.00 percentile latency (ns)": 2955167, "97.00 percentile latency (ns)": 2961797, "99.00 percentile latency (ns)": 2973997, "99.90 percentile latency (ns)": 3777410, "Max latency (ns)": 10647080, "Mean latency (ns)": 2904316, "Min duration satisfied": "Yes", "Min latency (ns)": 2749447, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 340.24, "QPS w/o loadgen overhead": 344.32, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 2.945716, "characteristics.90th_percentile_latency_ns": 2945716.0, "characteristics.90th_percentile_latency_s": 0.002945716, "characteristics.90th_percentile_latency_us": 2945.716, "characteristics.mAP": 20.118, "ck_system": "A30x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 326.413, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "4f4eab1c046584b0", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 431877, "90.00 percentile latency (ns)": 436816, "90th percentile latency (ns)": 436816, "95.00 percentile latency (ns)": 439442, "97.00 percentile latency (ns)": 441585, "99.00 percentile latency (ns)": 444942, "99.90 percentile latency (ns)": 511136, "Max latency (ns)": 9066898, "Mean latency (ns)": 432702, "Min duration satisfied": "Yes", "Min latency (ns)": 391881, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2290.68, "QPS w/o loadgen overhead": 2311.06, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.436816, "characteristics.90th_percentile_latency_ns": 436816.0, "characteristics.90th_percentile_latency_s": 0.000436816, "characteristics.90th_percentile_latency_us": 436.816, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "39564071a7c47ee8", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8135817, "90.00 percentile latency (ns)": 8156468, "90th percentile latency (ns)": 8156468, "95.00 percentile latency (ns)": 8162329, "97.00 percentile latency (ns)": 8165995, "99.00 percentile latency (ns)": 8173680, "99.90 percentile latency (ns)": 8402994, "Max latency (ns)": 13114726, "Mean latency (ns)": 8137825, "Min duration satisfied": "Yes", "Min latency (ns)": 8098368, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 122.8, "QPS w/o loadgen overhead": 122.88, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 8.156468, "characteristics.90th_percentile_latency_ns": 8156468.0, "characteristics.90th_percentile_latency_s": 0.008156468, "characteristics.90th_percentile_latency_us": 8156.468, "characteristics.mAP": 20.118, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "system_name": "NVIDIA DGX A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "746b9bd0678fb78d", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 396470, "90.00 percentile latency (ns)": 405140, "90th percentile latency (ns)": 405140, "95.00 percentile latency (ns)": 411911, "97.00 percentile latency (ns)": 416400, "99.00 percentile latency (ns)": 425650, "99.90 percentile latency (ns)": 443691, "Max latency (ns)": 10296242, "Mean latency (ns)": 398142, "Min duration satisfied": "Yes", "Min latency (ns)": 375070, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2458.85, "QPS w/o loadgen overhead": 2511.67, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.6gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.40514, "characteristics.90th_percentile_latency_ns": 405140.0, "characteristics.90th_percentile_latency_s": 0.00040514, "characteristics.90th_percentile_latency_us": 405.14, "characteristics.mAP": 22.914, "ck_system": "A30-MIG_1x1g.6gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.6gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.6gb_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30-MIG-1x1g.6gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1877.97, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "80a948cc143bfb2e", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8388220, "90.00 percentile latency (ns)": 8419851, "90th percentile latency (ns)": 8419851, "95.00 percentile latency (ns)": 8427841, "97.00 percentile latency (ns)": 8433261, "99.00 percentile latency (ns)": 8445201, "99.90 percentile latency (ns)": 9500001, "Max latency (ns)": 12913402, "Mean latency (ns)": 8387976, "Min duration satisfied": "Yes", "Min latency (ns)": 8146321, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 118.77, "QPS w/o loadgen overhead": 119.22, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.6gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 8.419851, "characteristics.90th_percentile_latency_ns": 8419851.0, "characteristics.90th_percentile_latency_s": 0.008419851, "characteristics.90th_percentile_latency_us": 8419.851, "characteristics.mAP": 20.118, "ck_system": "A30-MIG_1x1g.6gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.6gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.6gb_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30-MIG-1x1g.6gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 107.693, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "014f4319b9841028", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 309178, "90.00 percentile latency (ns)": 318549, "90th percentile latency (ns)": 318549, "95.00 percentile latency (ns)": 322748, "97.00 percentile latency (ns)": 328718, "99.00 percentile latency (ns)": 341778, "99.90 percentile latency (ns)": 369988, "Max latency (ns)": 7302920, "Mean latency (ns)": 310816, "Min duration satisfied": "Yes", "Min latency (ns)": 268829, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3137.29, "QPS w/o loadgen overhead": 3217.34, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.318549, "characteristics.90th_percentile_latency_ns": 318549.0, "characteristics.90th_percentile_latency_s": 0.000318549, "characteristics.90th_percentile_latency_us": 318.549, "characteristics.mAP": 22.914, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "bab45bcadee59d18", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1886800, "90.00 percentile latency (ns)": 1922090, "90th percentile latency (ns)": 1922090, "95.00 percentile latency (ns)": 1934330, "97.00 percentile latency (ns)": 1942770, "99.00 percentile latency (ns)": 1962430, "99.90 percentile latency (ns)": 2955524, "Max latency (ns)": 8465776, "Mean latency (ns)": 1892476, "Min duration satisfied": "Yes", "Min latency (ns)": 1818840, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 520.51, "QPS w/o loadgen overhead": 528.41, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.92209, "characteristics.90th_percentile_latency_ns": 1922090.0, "characteristics.90th_percentile_latency_s": 0.00192209, "characteristics.90th_percentile_latency_us": 1922.09, "characteristics.mAP": 20.118, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 8.0.1, CUDA 11.3", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 20.04.4", "other_hardware": "", "other_software_stack": "TensorRT 8.0.1, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "46fc91a08142066a", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1342337, "90.00 percentile latency (ns)": 1358271, "90th percentile latency (ns)": 1358271, "95.00 percentile latency (ns)": 1366016, "97.00 percentile latency (ns)": 1373024, "99.00 percentile latency (ns)": 1395232, "99.90 percentile latency (ns)": 3065767, "Max latency (ns)": 38608473, "Mean latency (ns)": 1349043, "Min duration satisfied": "Yes", "Min latency (ns)": 1280733, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 732.56, "QPS w/o loadgen overhead": 741.27, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.358271, "characteristics.90th_percentile_latency_ns": 1358271.0, "characteristics.90th_percentile_latency_s": 0.001358271, "characteristics.90th_percentile_latency_us": 1358.271, "characteristics.mAP": 22.914, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 500, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "325ff9bae4ca7810", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 41932913, "90.00 percentile latency (ns)": 42001892, "90th percentile latency (ns)": 42001892, "95.00 percentile latency (ns)": 42023173, "97.00 percentile latency (ns)": 42042726, "99.00 percentile latency (ns)": 42090822, "99.90 percentile latency (ns)": 44052528, "Max latency (ns)": 65151866, "Mean latency (ns)": 41945601, "Min duration satisfied": "Yes", "Min latency (ns)": 41730463, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 23.83, "QPS w/o loadgen overhead": 23.84, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 42.001892, "characteristics.90th_percentile_latency_ns": 42001892.0, "characteristics.90th_percentile_latency_s": 0.042001892, "characteristics.90th_percentile_latency_us": 42001.892, "characteristics.mAP": 20.113, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline scenario", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.1, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04", "other_hardware": "", "other_software_stack": "JetPack 4.6, TensorRT 8.0.1, CUDA 10.2, cuDNN 8.2.3, DALI 0.31.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 1624344308455410291, "retraining": "No", "sample_index_rng_seed": 517984244576520566, "samples_per_query": 1, "schedule_rng_seed": 10051496985653635065, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 16.9618, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "e9f92b4913a3390d", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 300339, "90.00 percentile latency (ns)": 312529, "90th percentile latency (ns)": 312529, "95.00 percentile latency (ns)": 314869, "97.00 percentile latency (ns)": 317019, "99.00 percentile latency (ns)": 325029, "99.90 percentile latency (ns)": 348939, "Max latency (ns)": 10382780, "Mean latency (ns)": 302894, "Min duration satisfied": "Yes", "Min latency (ns)": 265909, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3209.35, "QPS w/o loadgen overhead": 3301.49, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.312529, "characteristics.90th_percentile_latency_ns": 312529.0, "characteristics.90th_percentile_latency_s": 0.000312529, "characteristics.90th_percentile_latency_us": 312.529, "characteristics.mAP": 22.914, "ck_system": "A30x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2941.18, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "ba6849f49e8a9a53", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 2845650, "90.00 percentile latency (ns)": 2893420, "90th percentile latency (ns)": 2893420, "95.00 percentile latency (ns)": 2907740, "97.00 percentile latency (ns)": 2916882, "99.00 percentile latency (ns)": 2934970, "99.90 percentile latency (ns)": 3949807, "Max latency (ns)": 9374683, "Mean latency (ns)": 2846899, "Min duration satisfied": "Yes", "Min latency (ns)": 2693672, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 348.0, "QPS w/o loadgen overhead": 351.26, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 2.89342, "characteristics.90th_percentile_latency_ns": 2893420.0, "characteristics.90th_percentile_latency_s": 0.00289342, "characteristics.90th_percentile_latency_us": 2893.42, "characteristics.mAP": 20.111, "ck_system": "A30x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 326.413, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "3cc9d45b790a58cd", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1652632, "90.00 percentile latency (ns)": 1672792, "90th percentile latency (ns)": 1672792, "95.00 percentile latency (ns)": 1681431, "97.00 percentile latency (ns)": 1689816, "99.00 percentile latency (ns)": 1715897, "99.90 percentile latency (ns)": 2799205, "Max latency (ns)": 58255143, "Mean latency (ns)": 1659110, "Min duration satisfied": "Yes", "Min latency (ns)": 1598005, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 598.63, "QPS w/o loadgen overhead": 602.73, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.672792, "characteristics.90th_percentile_latency_ns": 1672792.0, "characteristics.90th_percentile_latency_s": 0.001672792, "characteristics.90th_percentile_latency_us": 1672.792, "characteristics.mAP": 22.914, "characteristics.power": 0.019098086120405512, "characteristics.power.normalized_per_core": 0.019098086120405512, "characteristics.power.normalized_per_processor": 0.019098086120405512, "ck_system": "Xavier_NX_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_MaxQ", "system_name": "NVIDIA Jetson Xavier NX (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 500, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "d8cd36744e4eb746", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 53629003, "90.00 percentile latency (ns)": 53735635, "90th percentile latency (ns)": 53735635, "95.00 percentile latency (ns)": 53769295, "97.00 percentile latency (ns)": 53792338, "99.00 percentile latency (ns)": 53853169, "99.90 percentile latency (ns)": 55669875, "Max latency (ns)": 69350736, "Mean latency (ns)": 53637919, "Min duration satisfied": "Yes", "Min latency (ns)": 53330121, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 18.64, "QPS w/o loadgen overhead": 18.64, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 53.735635, "characteristics.90th_percentile_latency_ns": 53735635.0, "characteristics.90th_percentile_latency_s": 0.053735635, "characteristics.90th_percentile_latency_us": 53735.635, "characteristics.mAP": 20.111, "characteristics.power": 0.7738311530357825, "characteristics.power.normalized_per_core": 0.7738311530357825, "characteristics.power.normalized_per_processor": 0.7738311530357825, "ck_system": "Xavier_NX_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/Xavier_NX_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT_MaxQ", "system_name": "NVIDIA Jetson Xavier NX (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 16.9618, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "e95e4256b5fcfda5", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 281188, "90.00 percentile latency (ns)": 288398, "90th percentile latency (ns)": 288398, "95.00 percentile latency (ns)": 294648, "97.00 percentile latency (ns)": 298887, "99.00 percentile latency (ns)": 315758, "99.90 percentile latency (ns)": 332507, "Max latency (ns)": 9493280, "Mean latency (ns)": 283824, "Min duration satisfied": "Yes", "Min latency (ns)": 255978, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3435.19, "QPS w/o loadgen overhead": 3523.31, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.288398, "characteristics.90th_percentile_latency_ns": 288398.0, "characteristics.90th_percentile_latency_s": 0.000288398, "characteristics.90th_percentile_latency_us": 288.398, "characteristics.mAP": 22.914, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "24dc09d5074fb2b7", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1966025, "90.00 percentile latency (ns)": 2010144, "90th percentile latency (ns)": 2010144, "95.00 percentile latency (ns)": 2024793, "97.00 percentile latency (ns)": 2034564, "99.00 percentile latency (ns)": 2057423, "99.90 percentile latency (ns)": 2582980, "Max latency (ns)": 10081459, "Mean latency (ns)": 1968049, "Min duration satisfied": "Yes", "Min latency (ns)": 1823005, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 504.88, "QPS w/o loadgen overhead": 508.12, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 2.010144, "characteristics.90th_percentile_latency_ns": 2010144.0, "characteristics.90th_percentile_latency_s": 0.002010144, "characteristics.90th_percentile_latency_us": 2010.144, "characteristics.mAP": 20.111, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "d3979a48aea99543", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1020501, "90.00 percentile latency (ns)": 1039319, "90th percentile latency (ns)": 1039319, "95.00 percentile latency (ns)": 1047255, "97.00 percentile latency (ns)": 1055863, "99.00 percentile latency (ns)": 1176223, "99.90 percentile latency (ns)": 1716891, "Max latency (ns)": 7636433, "Mean latency (ns)": 1025839, "Min duration satisfied": "Yes", "Min latency (ns)": 965139, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 963.66, "QPS w/o loadgen overhead": 974.81, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.039319, "characteristics.90th_percentile_latency_ns": 1039319.0, "characteristics.90th_percentile_latency_s": 0.001039319, "characteristics.90th_percentile_latency_us": 1039.319, "characteristics.mAP": 22.914, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 666.667, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "258d30441fa26c9a", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 25851694, "90.00 percentile latency (ns)": 25966677, "90th percentile latency (ns)": 25966677, "95.00 percentile latency (ns)": 26016858, "97.00 percentile latency (ns)": 26051639, "99.00 percentile latency (ns)": 26186560, "99.90 percentile latency (ns)": 26689435, "Max latency (ns)": 29957253, "Mean latency (ns)": 25865153, "Min duration satisfied": "Yes", "Min latency (ns)": 25684970, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 38.65, "QPS w/o loadgen overhead": 38.66, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 25.966677, "characteristics.90th_percentile_latency_ns": 25966677.0, "characteristics.90th_percentile_latency_s": 0.025966677, "characteristics.90th_percentile_latency_us": 25966.677, "characteristics.mAP": 20.111, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 33.9236, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "430a83c3401f8527", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 312121, "90.00 percentile latency (ns)": 317708, "90th percentile latency (ns)": 317708, "95.00 percentile latency (ns)": 321061, "97.00 percentile latency (ns)": 322946, "99.00 percentile latency (ns)": 326857, "99.90 percentile latency (ns)": 398026, "Max latency (ns)": 9673785, "Mean latency (ns)": 313840, "Min duration satisfied": "Yes", "Min latency (ns)": 252406, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3069.7, "QPS w/o loadgen overhead": 3186.34, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.317708, "characteristics.90th_percentile_latency_ns": 317708.0, "characteristics.90th_percentile_latency_s": 0.000317708, "characteristics.90th_percentile_latency_us": 317.708, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_edge", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_edge", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_edge", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "543aba873f1f1adc", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1738768, "90.00 percentile latency (ns)": 1781512, "90th percentile latency (ns)": 1781512, "95.00 percentile latency (ns)": 1819575, "97.00 percentile latency (ns)": 1828934, "99.00 percentile latency (ns)": 1850305, "99.90 percentile latency (ns)": 3893791, "Max latency (ns)": 8451842, "Mean latency (ns)": 1753406, "Min duration satisfied": "Yes", "Min latency (ns)": 1643575, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 567.07, "QPS w/o loadgen overhead": 570.32, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.781512, "characteristics.90th_percentile_latency_ns": 1781512.0, "characteristics.90th_percentile_latency_s": 0.001781512, "characteristics.90th_percentile_latency_us": 1781.512, "characteristics.mAP": 20.111, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_edge", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_edge", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_edge", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "532921dd4d008ebc", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 466705, "90.00 percentile latency (ns)": 473007, "90th percentile latency (ns)": 473007, "95.00 percentile latency (ns)": 475452, "97.00 percentile latency (ns)": 477516, "99.00 percentile latency (ns)": 481653, "99.90 percentile latency (ns)": 574066, "Max latency (ns)": 5587408, "Mean latency (ns)": 467406, "Min duration satisfied": "Yes", "Min latency (ns)": 426840, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2116.89, "QPS w/o loadgen overhead": 2139.47, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.473007, "characteristics.90th_percentile_latency_ns": 473007.0, "characteristics.90th_percentile_latency_s": 0.000473007, "characteristics.90th_percentile_latency_us": 473.007, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "system_name": "NVIDIA DGX-A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "2efd847daafca5d7", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8175440, "90.00 percentile latency (ns)": 8196239, "90th percentile latency (ns)": 8196239, "95.00 percentile latency (ns)": 8203122, "97.00 percentile latency (ns)": 8207972, "99.00 percentile latency (ns)": 8240703, "99.90 percentile latency (ns)": 8309421, "Max latency (ns)": 10298141, "Mean latency (ns)": 8175906, "Min duration satisfied": "Yes", "Min latency (ns)": 8113735, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 122.21, "QPS w/o loadgen overhead": 122.31, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 8.196239, "characteristics.90th_percentile_latency_ns": 8196239.0, "characteristics.90th_percentile_latency_s": 0.008196239, "characteristics.90th_percentile_latency_us": 8196.239, "characteristics.mAP": 20.111, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT_Triton", "system_name": "NVIDIA DGX-A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "50d5b8b1154809b8", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1770292, "90.00 percentile latency (ns)": 1786741, "90th percentile latency (ns)": 1786741, "95.00 percentile latency (ns)": 1794965, "97.00 percentile latency (ns)": 1803413, "99.00 percentile latency (ns)": 1827991, "99.90 percentile latency (ns)": 2327279, "Max latency (ns)": 20900995, "Mean latency (ns)": 1772753, "Min duration satisfied": "Yes", "Min latency (ns)": 1704369, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 560.62, "QPS w/o loadgen overhead": 564.09, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.786741, "characteristics.90th_percentile_latency_ns": 1786741.0, "characteristics.90th_percentile_latency_s": 0.001786741, "characteristics.90th_percentile_latency_us": 1786.741, "characteristics.mAP": 22.914, "characteristics.power": 0.024277169718655702, "characteristics.power.normalized_per_core": 0.024277169718655702, "characteristics.power.normalized_per_processor": 0.024277169718655702, "ck_system": "AGX_Xavier_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_MaxQ", "system_name": "NVIDIA Jetson AGX Xavier 32GB (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 666.667, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "1c157626f445ad80", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 50622951, "90.00 percentile latency (ns)": 50699778, "90th percentile latency (ns)": 50699778, "95.00 percentile latency (ns)": 50723543, "97.00 percentile latency (ns)": 50742062, "99.00 percentile latency (ns)": 50816173, "99.90 percentile latency (ns)": 51414117, "Max latency (ns)": 72170628, "Mean latency (ns)": 50630808, "Min duration satisfied": "Yes", "Min latency (ns)": 50409091, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 19.75, "QPS w/o loadgen overhead": 19.75, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 50.699778, "characteristics.90th_percentile_latency_ns": 50699778.0, "characteristics.90th_percentile_latency_s": 0.050699778, "characteristics.90th_percentile_latency_us": 50699.778, "characteristics.mAP": 20.111, "characteristics.power": 0.8885286888290141, "characteristics.power.normalized_per_core": 0.8885286888290141, "characteristics.power.normalized_per_processor": 0.8885286888290141, "ck_system": "AGX_Xavier_TRT_MaxQ", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "32 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT_MaxQ", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT_MaxQ", "system_name": "NVIDIA Jetson AGX Xavier 32GB (MaxQ, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 33.9236, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "59e5b531b486cc48", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 499002, "90.00 percentile latency (ns)": 510822, "90th percentile latency (ns)": 510822, "95.00 percentile latency (ns)": 526242, "97.00 percentile latency (ns)": 529311, "99.00 percentile latency (ns)": 534762, "99.90 percentile latency (ns)": 578981, "Max latency (ns)": 5150919, "Mean latency (ns)": 501420, "Min duration satisfied": "Yes", "Min latency (ns)": 463283, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1910.63, "QPS w/o loadgen overhead": 1994.34, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.3gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.510822, "characteristics.90th_percentile_latency_ns": 510822.0, "characteristics.90th_percentile_latency_s": 0.000510822, "characteristics.90th_percentile_latency_us": 510.822, "characteristics.mAP": 22.914, "ck_system": "A30-MIG_1x1g.3gb_TRT", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.3gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.3gb_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1951.22, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "fea411fa4f784425", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 9035974, "90.00 percentile latency (ns)": 9065025, "90th percentile latency (ns)": 9065025, "95.00 percentile latency (ns)": 9073144, "97.00 percentile latency (ns)": 9077894, "99.00 percentile latency (ns)": 9088613, "99.90 percentile latency (ns)": 9344510, "Max latency (ns)": 13087720, "Mean latency (ns)": 9034265, "Min duration satisfied": "Yes", "Min latency (ns)": 8838087, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 110.2, "QPS w/o loadgen overhead": 110.69, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.3gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 9.065025, "characteristics.90th_percentile_latency_ns": 9065025.0, "characteristics.90th_percentile_latency_s": 0.009065025, "characteristics.90th_percentile_latency_us": 9065.025, "characteristics.mAP": 20.111, "ck_system": "A30-MIG_1x1g.3gb_TRT", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.3gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.3gb_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 110.136, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "69e7741d6989a6a2", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 303355, "90.00 percentile latency (ns)": 309527, "90th percentile latency (ns)": 309527, "95.00 percentile latency (ns)": 311351, "97.00 percentile latency (ns)": 312538, "99.00 percentile latency (ns)": 315038, "99.90 percentile latency (ns)": 321678, "Max latency (ns)": 10871518, "Mean latency (ns)": 304148, "Min duration satisfied": "Yes", "Min latency (ns)": 285529, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3201.23, "QPS w/o loadgen overhead": 3287.87, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.309527, "characteristics.90th_percentile_latency_ns": 309527.0, "characteristics.90th_percentile_latency_s": 0.000309527, "characteristics.90th_percentile_latency_us": 309.527, "characteristics.mAP": 22.914, "ck_system": "A10x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A10x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT_Triton", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2680.97, "task": "object detection", "task2": "object detection", "total_cores": 56, "uid": "5e18366c1c0742b5", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 4190013, "90.00 percentile latency (ns)": 4256297, "90th percentile latency (ns)": 4256297, "95.00 percentile latency (ns)": 4272539, "97.00 percentile latency (ns)": 4282340, "99.00 percentile latency (ns)": 4302319, "99.90 percentile latency (ns)": 4381562, "Max latency (ns)": 16572420, "Mean latency (ns)": 4191930, "Min duration satisfied": "Yes", "Min latency (ns)": 3715198, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 234.48, "QPS w/o loadgen overhead": 238.55, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 4.256297, "characteristics.90th_percentile_latency_ns": 4256297.0, "characteristics.90th_percentile_latency_s": 0.004256297, "characteristics.90th_percentile_latency_us": 4256.297, "characteristics.mAP": 20.111, "ck_system": "A10x1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A10x1_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT_Triton", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 228.833, "task": "object detection", "task2": "object detection", "total_cores": 56, "uid": "0744ec8641430ee3", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 248676, "90.00 percentile latency (ns)": 255810, "90th percentile latency (ns)": 255810, "95.00 percentile latency (ns)": 257313, "97.00 percentile latency (ns)": 258976, "99.00 percentile latency (ns)": 262642, "99.90 percentile latency (ns)": 284183, "Max latency (ns)": 5792742, "Mean latency (ns)": 250256, "Min duration satisfied": "Yes", "Min latency (ns)": 234700, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 3940.06, "QPS w/o loadgen overhead": 3995.91, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.25581, "characteristics.90th_percentile_latency_ns": 255810.0, "characteristics.90th_percentile_latency_s": 0.00025581, "characteristics.90th_percentile_latency_us": 255.81, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "6e5540dab92e7446", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1668078, "90.00 percentile latency (ns)": 1686353, "90th percentile latency (ns)": 1686353, "95.00 percentile latency (ns)": 1693305, "97.00 percentile latency (ns)": 1699005, "99.00 percentile latency (ns)": 1758137, "99.90 percentile latency (ns)": 1865388, "Max latency (ns)": 6674486, "Mean latency (ns)": 1669788, "Min duration satisfied": "Yes", "Min latency (ns)": 1607094, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 596.58, "QPS w/o loadgen overhead": 598.88, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.686353, "characteristics.90th_percentile_latency_ns": 1686353.0, "characteristics.90th_percentile_latency_s": 0.001686353, "characteristics.90th_percentile_latency_us": 1686.353, "characteristics.mAP": 20.111, "ck_system": "DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0, Triton 21.02; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GBx1_TRT_Triton_edge", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "fd4ceb964a05b254", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 336463, "90.00 percentile latency (ns)": 348363, "90th percentile latency (ns)": 348363, "95.00 percentile latency (ns)": 351466, "97.00 percentile latency (ns)": 354145, "99.00 percentile latency (ns)": 359216, "99.90 percentile latency (ns)": 404646, "Max latency (ns)": 4221831, "Mean latency (ns)": 337325, "Min duration satisfied": "Yes", "Min latency (ns)": 296688, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2906.18, "QPS w/o loadgen overhead": 2964.5, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.348363, "characteristics.90th_percentile_latency_ns": 348363.0, "characteristics.90th_percentile_latency_s": 0.000348363, "characteristics.90th_percentile_latency_us": 348.363, "characteristics.mAP": 22.914, "ck_system": "A10x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A10x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2680.97, "task": "object detection", "task2": "object detection", "total_cores": 56, "uid": "e4d09afe7c682069", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 4055974, "90.00 percentile latency (ns)": 4101104, "90th percentile latency (ns)": 4101104, "95.00 percentile latency (ns)": 4114983, "97.00 percentile latency (ns)": 4125249, "99.00 percentile latency (ns)": 4150483, "99.90 percentile latency (ns)": 4246315, "Max latency (ns)": 6507540, "Mean latency (ns)": 4052771, "Min duration satisfied": "Yes", "Min latency (ns)": 3534147, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 242.94, "QPS w/o loadgen overhead": 246.74, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA A10", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 4.101104, "characteristics.90th_percentile_latency_ns": 4101104.0, "characteristics.90th_percentile_latency_s": 0.004101104, "characteristics.90th_percentile_latency_us": 4101.104, "characteristics.mAP": 20.111, "ck_system": "A10x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A10x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A10x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x A10, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 228.833, "task": "object detection", "task2": "object detection", "total_cores": 56, "uid": "398bf7497048d9d6", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 329639, "90.00 percentile latency (ns)": 339259, "90th percentile latency (ns)": 339259, "95.00 percentile latency (ns)": 345118, "97.00 percentile latency (ns)": 357189, "99.00 percentile latency (ns)": 369219, "99.90 percentile latency (ns)": 459969, "Max latency (ns)": 11195766, "Mean latency (ns)": 331914, "Min duration satisfied": "Yes", "Min latency (ns)": 280539, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2924.84, "QPS w/o loadgen overhead": 3012.83, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.339259, "characteristics.90th_percentile_latency_ns": 339259.0, "characteristics.90th_percentile_latency_s": 0.000339259, "characteristics.90th_percentile_latency_us": 339.259, "characteristics.mAP": 22.914, "ck_system": "A30x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2941.18, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "ab46df815e51bf8c", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 2957431, "90.00 percentile latency (ns)": 2990452, "90th percentile latency (ns)": 2990452, "95.00 percentile latency (ns)": 2998962, "97.00 percentile latency (ns)": 3004732, "99.00 percentile latency (ns)": 3016610, "99.90 percentile latency (ns)": 3405521, "Max latency (ns)": 10542223, "Mean latency (ns)": 2952717, "Min duration satisfied": "Yes", "Min latency (ns)": 2728860, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 334.44, "QPS w/o loadgen overhead": 338.67, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 2.990452, "characteristics.90th_percentile_latency_ns": 2990452.0, "characteristics.90th_percentile_latency_s": 0.002990452, "characteristics.90th_percentile_latency_us": 2990.452, "characteristics.mAP": 20.111, "ck_system": "A30x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30x1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30x1_TRT", "system_name": "Gigabyte G482-Z54 (1x A30, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 326.413, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "8d8b250a16429eb2", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 517692, "90.00 percentile latency (ns)": 531602, "90th percentile latency (ns)": 531602, "95.00 percentile latency (ns)": 539902, "97.00 percentile latency (ns)": 545141, "99.00 percentile latency (ns)": 552742, "99.90 percentile latency (ns)": 621350, "Max latency (ns)": 11099185, "Mean latency (ns)": 520418, "Min duration satisfied": "Yes", "Min latency (ns)": 479892, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 1886.28, "QPS w/o loadgen overhead": 1921.53, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.3gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.531602, "characteristics.90th_percentile_latency_ns": 531602.0, "characteristics.90th_percentile_latency_s": 0.000531602, "characteristics.90th_percentile_latency_us": 531.602, "characteristics.mAP": 22.914, "ck_system": "A30-MIG_1x1g.3gb_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.3gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0, Triton 21.02; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.3gb_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1877.97, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "44d4234d9da0edb7", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 9253002, "90.00 percentile latency (ns)": 9287861, "90th percentile latency (ns)": 9287861, "95.00 percentile latency (ns)": 9302952, "97.00 percentile latency (ns)": 9314091, "99.00 percentile latency (ns)": 9336221, "99.90 percentile latency (ns)": 9444479, "Max latency (ns)": 19753415, "Mean latency (ns)": 9250676, "Min duration satisfied": "Yes", "Min latency (ns)": 8926838, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 107.67, "QPS w/o loadgen overhead": 108.1, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "24 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A30 (1x1g.3gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 9.287861, "characteristics.90th_percentile_latency_ns": 9287861.0, "characteristics.90th_percentile_latency_s": 0.009287861, "characteristics.90th_percentile_latency_us": 9287.861, "characteristics.mAP": 20.111, "ck_system": "A30-MIG_1x1g.3gb_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A30-MIG_1x1g.3gb_TRT_Triton", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.46, DALI 0.30.0, Triton 21.02; GCC 7.5.0; Python 3.7.10", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "preview", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A30-MIG_1x1g.3gb_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 107.693, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "4eea8bbf22c27cc7", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 448311, "90.00 percentile latency (ns)": 453460, "90th percentile latency (ns)": 453460, "95.00 percentile latency (ns)": 456456, "97.00 percentile latency (ns)": 458660, "99.00 percentile latency (ns)": 462277, "99.90 percentile latency (ns)": 533751, "Max latency (ns)": 3680531, "Mean latency (ns)": 449265, "Min duration satisfied": "Yes", "Min latency (ns)": 403417, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2203.29, "QPS w/o loadgen overhead": 2225.86, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.45346, "characteristics.90th_percentile_latency_ns": 453460.0, "characteristics.90th_percentile_latency_s": 0.00045346, "characteristics.90th_percentile_latency_us": 453.46, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "system_name": "NVIDIA DGX-A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "4cb454c77acf438c", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8159772, "90.00 percentile latency (ns)": 8179258, "90th percentile latency (ns)": 8179258, "95.00 percentile latency (ns)": 8184838, "97.00 percentile latency (ns)": 8188575, "99.00 percentile latency (ns)": 8195408, "99.90 percentile latency (ns)": 8239199, "Max latency (ns)": 11557763, "Mean latency (ns)": 8159669, "Min duration satisfied": "Yes", "Min latency (ns)": 8072567, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 122.47, "QPS w/o loadgen overhead": 122.55, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "80 GB", "accelerator_memory_configuration": "HBM2e", "accelerator_model_name": "NVIDIA A100-SXM-80GB (1x1g.10gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 8.179258, "characteristics.90th_percentile_latency_ns": 8179258.0, "characteristics.90th_percentile_latency_s": 0.008179258, "characteristics.90th_percentile_latency_us": 8179.258, "characteristics.mAP": 20.111, "ck_system": "DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "2 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM-80GB-MIG_1x1g.10gb_TRT", "system_name": "NVIDIA DGX-A100 (1x A100-SXM-80GB-MIG-1x1g.10gb, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "8443da254304b987", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 324177, "90.00 percentile latency (ns)": 331458, "90th percentile latency (ns)": 331458, "95.00 percentile latency (ns)": 333987, "97.00 percentile latency (ns)": 336108, "99.00 percentile latency (ns)": 347407, "99.90 percentile latency (ns)": 640665, "Max latency (ns)": 12342601, "Mean latency (ns)": 326216, "Min duration satisfied": "Yes", "Min latency (ns)": 269369, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 2964.05, "QPS w/o loadgen overhead": 3065.45, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 0.331458, "characteristics.90th_percentile_latency_ns": 331458.0, "characteristics.90th_percentile_latency_s": 0.000331458, "characteristics.90th_percentile_latency_us": 331.458, "characteristics.mAP": 22.914, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "683fcc8346faa055", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1931124, "90.00 percentile latency (ns)": 1965664, "90th percentile latency (ns)": 1965664, "95.00 percentile latency (ns)": 1977415, "97.00 percentile latency (ns)": 1986334, "99.00 percentile latency (ns)": 2012284, "99.90 percentile latency (ns)": 3878329, "Max latency (ns)": 12876438, "Mean latency (ns)": 1934936, "Min duration satisfied": "Yes", "Min latency (ns)": 1806966, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 508.71, "QPS w/o loadgen overhead": 516.81, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40 GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe-40GB", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.965664, "characteristics.90th_percentile_latency_ns": 1965664.0, "characteristics.90th_percentile_latency_s": 0.001965664, "characteristics.90th_percentile_latency_us": 1965.664, "characteristics.mAP": 20.111, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2.3, CUDA 11.1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "TensorRT 7.2.3, CUDA 11.1, cuDNN 8.1.1, Driver 460.32.03, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z54 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "48f79980a4d589d9", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1409546, "90.00 percentile latency (ns)": 1429581, "90th percentile latency (ns)": 1429581, "95.00 percentile latency (ns)": 1437705, "97.00 percentile latency (ns)": 1444393, "99.00 percentile latency (ns)": 1467439, "99.90 percentile latency (ns)": 2473348, "Max latency (ns)": 42138492, "Mean latency (ns)": 1417070, "Min duration satisfied": "Yes", "Min latency (ns)": 1372551, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 701.9, "QPS w/o loadgen overhead": 705.68, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 1.429581, "characteristics.90th_percentile_latency_ns": 1429581.0, "characteristics.90th_percentile_latency_s": 0.001429581, "characteristics.90th_percentile_latency_us": 1429.581, "characteristics.mAP": 22.914, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 1024, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 500, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "7e78620f87e84c5f", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 42829342, "90.00 percentile latency (ns)": 42900563, "90th percentile latency (ns)": 42900563, "95.00 percentile latency (ns)": 42925823, "97.00 percentile latency (ns)": 42948011, "99.00 percentile latency (ns)": 43037357, "99.90 percentile latency (ns)": 45114950, "Max latency (ns)": 58876686, "Mean latency (ns)": 42842929, "Min duration satisfied": "Yes", "Min latency (ns)": 42658066, "Min queries satisfied": "Yes", "Mode": "PerformanceOnly", "QPS w/ loadgen overhead": 23.33, "QPS w/o loadgen overhead": 23.34, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "boot_firmware_version": "", "characteristics.90th_percentile_latency_ms": 42.900563, "characteristics.90th_percentile_latency_ns": 42900563.0, "characteristics.90th_percentile_latency_s": 0.042900563, "characteristics.90th_percentile_latency_us": 42900.563, "characteristics.mAP": 20.111, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "disk_controllers": "", "disk_drives": "", "division": "closed", "filesystem": "", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2", "host_memory_capacity": "8 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32 GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "management_firmware_version": "", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 600000, "min_query_count": 1024, "mlperf_version": 1.0, "network_speed_mbit": "", "nics_enabled_connected": "", "nics_enabled_firmware": "", "nics_enabled_os": "", "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v1.0/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "number_of_type_nics_installed": "", "operating_system": "Ubuntu 18.04.4", "other_hardware": "", "other_software_stack": "21.03 Jetson CUDA-X AI Developer Preview, TensorRT 7.2.3, CUDA 10.2, cuDNN 8.0.0, DALI 0.30.0", "performance_issue_same": 0, "performance_issue_same_index": 0, "performance_issue_unique": 0, "performance_sample_count": 64, "power_management": "", "power_supply_details": "", "power_supply_quantity_and_rating_watts": "", "print_timestamps": 0, "problem": false, "qsl_rng_seed": 7322528924094909334, "retraining": "N", "sample_index_rng_seed": 1570999273408051088, "samples_per_query": 1, "schedule_rng_seed": 3507442325620259414, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 16.9618, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "4f4fca8280aa1fe2", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 307477, "90.00 percentile latency (ns)": 312567, "90th percentile latency (ns)": 312567, "95.00 percentile latency (ns)": 315271, "97.00 percentile latency (ns)": 317095, "99.00 percentile latency (ns)": 320612, "99.90 percentile latency (ns)": 428964, "Max latency (ns)": 5772740, "Mean latency (ns)": 308167, "Min duration satisfied": "Yes", "Min latency (ns)": 269236, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 3186.25, "QPS w/o loadgen overhead": 3244.99, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 0.312567, "characteristics.90th_percentile_latency_ns": 312567.0, "characteristics.90th_percentile_latency_s": 0.000312567, "characteristics.90th_percentile_latency_us": 312.567, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM4x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 1024, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1_TRT", "system_name": "NVIDIA DGX-A100 (1x A100-SXM4, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "68cd9da225521331", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1734774, "90.00 percentile latency (ns)": 1761845, "90th percentile latency (ns)": 1761845, "95.00 percentile latency (ns)": 1768899, "97.00 percentile latency (ns)": 1774118, "99.00 percentile latency (ns)": 1785289, "99.90 percentile latency (ns)": 4360620, "Max latency (ns)": 6205441, "Mean latency (ns)": 1737038, "Min duration satisfied": "Yes", "Min latency (ns)": 1664172, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 573.5, "QPS w/o loadgen overhead": 575.69, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 1.761845, "characteristics.90th_percentile_latency_ns": 1761845.0, "characteristics.90th_percentile_latency_s": 0.001761845, "characteristics.90th_percentile_latency_us": 1761.845, "characteristics.mAP": 20.111, "ck_system": "DGX-A100_A100-SXM4x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 64, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1_TRT", "system_name": "NVIDIA DGX-A100 (1x A100-SXM4, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "23bdb281c651f5f9", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 520606, "90.00 percentile latency (ns)": 527078, "90th percentile latency (ns)": 527078, "95.00 percentile latency (ns)": 530125, "97.00 percentile latency (ns)": 532238, "99.00 percentile latency (ns)": 537338, "99.90 percentile latency (ns)": 647645, "Max latency (ns)": 3043990, "Mean latency (ns)": 520316, "Min duration satisfied": "Yes", "Min latency (ns)": 473438, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 1894.82, "QPS w/o loadgen overhead": 1921.91, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "5GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4 (1x1g.5gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 0.527078, "characteristics.90th_percentile_latency_ns": 527078.0, "characteristics.90th_percentile_latency_s": 0.000527078, "characteristics.90th_percentile_latency_us": 527.078, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 1024, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "3fc4d3e965d0e63b", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8485035, "90.00 percentile latency (ns)": 8514429, "90th percentile latency (ns)": 8514429, "95.00 percentile latency (ns)": 8522885, "97.00 percentile latency (ns)": 8530891, "99.00 percentile latency (ns)": 8588288, "99.90 percentile latency (ns)": 8645496, "Max latency (ns)": 9580489, "Mean latency (ns)": 8482367, "Min duration satisfied": "Yes", "Min latency (ns)": 8383955, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 117.78, "QPS w/o loadgen overhead": 117.89, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "5GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4 (1x1g.5gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 8.514429, "characteristics.90th_percentile_latency_ns": 8514429.0, "characteristics.90th_percentile_latency_s": 0.008514429, "characteristics.90th_percentile_latency_us": 8514.429, "characteristics.mAP": 20.111, "ck_system": "DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 64, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "4576f7274c62a144", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 363044, "90.00 percentile latency (ns)": 372694, "90th percentile latency (ns)": 372694, "95.00 percentile latency (ns)": 377774, "97.00 percentile latency (ns)": 383934, "99.00 percentile latency (ns)": 402303, "99.90 percentile latency (ns)": 501221, "Max latency (ns)": 3558550, "Mean latency (ns)": 365120, "Min duration satisfied": "Yes", "Min latency (ns)": 325134, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 2662.0, "QPS w/o loadgen overhead": 2738.83, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 0.372694, "characteristics.90th_percentile_latency_ns": 372694.0, "characteristics.90th_percentile_latency_s": 0.000372694, "characteristics.90th_percentile_latency_us": 372.694, "characteristics.mAP": 22.914, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 1024, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z52 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "d4fa902084606a3a", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1994636, "90.00 percentile latency (ns)": 2038186, "90th percentile latency (ns)": 2038186, "95.00 percentile latency (ns)": 2051865, "97.00 percentile latency (ns)": 2061206, "99.00 percentile latency (ns)": 2079515, "99.90 percentile latency (ns)": 2676655, "Max latency (ns)": 5411498, "Mean latency (ns)": 1996564, "Min duration satisfied": "Yes", "Min latency (ns)": 1877858, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 492.76, "QPS w/o loadgen overhead": 500.86, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 2.038186, "characteristics.90th_percentile_latency_ns": 2038186.0, "characteristics.90th_percentile_latency_s": 0.002038186, "characteristics.90th_percentile_latency_us": 2038.186, "characteristics.mAP": 20.111, "ck_system": "A100-PCIex1_TRT_Triton", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 64, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT_Triton", "system_name": "Gigabyte G482-Z52 (1x A100-PCIe, TensorRT, Triton)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "088ec78d040a499c", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1215382, "90.00 percentile latency (ns)": 1245207, "90th percentile latency (ns)": 1245207, "95.00 percentile latency (ns)": 1259960, "97.00 percentile latency (ns)": 1271576, "99.00 percentile latency (ns)": 1302362, "99.90 percentile latency (ns)": 1460449, "Max latency (ns)": 8225613, "Mean latency (ns)": 1220073, "Min duration satisfied": "Yes", "Min latency (ns)": 1164660, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 811.66, "QPS w/o loadgen overhead": 819.62, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 1.245207, "characteristics.90th_percentile_latency_ns": 1245207.0, "characteristics.90th_percentile_latency_s": 0.001245207, "characteristics.90th_percentile_latency_us": 1245.207, "characteristics.mAP": 22.9, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2", "host_memory_capacity": "32GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 1024, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 616.903, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "b63ce834d0994f28", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 28389887, "90.00 percentile latency (ns)": 28531845, "90th percentile latency (ns)": 28531845, "95.00 percentile latency (ns)": 28577734, "97.00 percentile latency (ns)": 28610953, "99.00 percentile latency (ns)": 28704172, "99.90 percentile latency (ns)": 28920982, "Max latency (ns)": 29715930, "Mean latency (ns)": 28394994, "Min duration satisfied": "Yes", "Min latency (ns)": 28105874, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 35.2, "QPS w/o loadgen overhead": 35.22, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA AGX Xavier", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 28.531845, "characteristics.90th_percentile_latency_ns": 28531845.0, "characteristics.90th_percentile_latency_s": 0.028531845, "characteristics.90th_percentile_latency_us": 28531.845, "characteristics.mAP": 20.111, "ck_system": "AGX_Xavier_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2", "host_memory_capacity": "32GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 8, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32GB", "host_storage_type": "eMMC 5.1", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/AGX_Xavier_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 64, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/AGX_Xavier_TRT", "system_name": "NVIDIA Jetson AGX Xavier 32GB (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 33.9236, "task": "object detection", "task2": "object detection", "total_cores": 8, "uid": "4da192a0af53031d", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 327234, "90.00 percentile latency (ns)": 331863, "90th percentile latency (ns)": 331863, "95.00 percentile latency (ns)": 335429, "97.00 percentile latency (ns)": 337273, "99.00 percentile latency (ns)": 340348, "99.90 percentile latency (ns)": 453251, "Max latency (ns)": 1851242, "Mean latency (ns)": 328407, "Min duration satisfied": "Yes", "Min latency (ns)": 280937, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 2993.33, "QPS w/o loadgen overhead": 3045.0, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 0.331863, "characteristics.90th_percentile_latency_ns": 331863.0, "characteristics.90th_percentile_latency_s": 0.000331863, "characteristics.90th_percentile_latency_us": 331.863, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM4x1_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 1024, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "9d4accac8d0a7809", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1911787, "90.00 percentile latency (ns)": 1945280, "90th percentile latency (ns)": 1945280, "95.00 percentile latency (ns)": 2044255, "97.00 percentile latency (ns)": 2081535, "99.00 percentile latency (ns)": 2125978, "99.90 percentile latency (ns)": 2281820, "Max latency (ns)": 5435597, "Mean latency (ns)": 1918644, "Min duration satisfied": "Yes", "Min latency (ns)": 1838209, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 520.06, "QPS w/o loadgen overhead": 521.2, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 1.94528, "characteristics.90th_percentile_latency_ns": 1945280.0, "characteristics.90th_percentile_latency_s": 0.00194528, "characteristics.90th_percentile_latency_us": 1945.28, "characteristics.mAP": 20.111, "ck_system": "DGX-A100_A100-SXM4x1_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 64, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "c2f447ac83441fa8", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 483467, "90.00 percentile latency (ns)": 489248, "90th percentile latency (ns)": 489248, "95.00 percentile latency (ns)": 492013, "97.00 percentile latency (ns)": 493956, "99.00 percentile latency (ns)": 497824, "99.90 percentile latency (ns)": 612960, "Max latency (ns)": 2621247, "Mean latency (ns)": 484784, "Min duration satisfied": "Yes", "Min latency (ns)": 444413, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 2038.02, "QPS w/o loadgen overhead": 2062.77, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "5GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4 (1x1g.5gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 0.489248, "characteristics.90th_percentile_latency_ns": 489248.0, "characteristics.90th_percentile_latency_s": 0.000489248, "characteristics.90th_percentile_latency_us": 489.248, "characteristics.mAP": 22.914, "ck_system": "DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 1024, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "45bfb05ae977cdd5", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8313944, "90.00 percentile latency (ns)": 8341516, "90th percentile latency (ns)": 8341516, "95.00 percentile latency (ns)": 8348919, "97.00 percentile latency (ns)": 8352807, "99.00 percentile latency (ns)": 8361282, "99.90 percentile latency (ns)": 8482410, "Max latency (ns)": 10239284, "Mean latency (ns)": 8310901, "Min duration satisfied": "Yes", "Min latency (ns)": 8227322, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 120.21, "QPS w/o loadgen overhead": 120.32, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "5GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-SXM4 (1x1g.5gb MIG)", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 8.341516, "characteristics.90th_percentile_latency_ns": 8341516.0, "characteristics.90th_percentile_latency_s": 0.008341516, "characteristics.90th_percentile_latency_us": 8341.516, "characteristics.mAP": 20.111, "ck_system": "DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "1 TB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "15 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 64, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DGX-A100_A100-SXM4x1-MIG_1x1g.5gb_TRT", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "67d480884d91e7eb", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 597111, "90.00 percentile latency (ns)": 613363, "90th percentile latency (ns)": 613363, "95.00 percentile latency (ns)": 617483, "97.00 percentile latency (ns)": 620640, "99.00 percentile latency (ns)": 628217, "99.90 percentile latency (ns)": 720898, "Max latency (ns)": 8948440, "Mean latency (ns)": 596057, "Min duration satisfied": "Yes", "Min latency (ns)": 555004, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 1645.81, "QPS w/o loadgen overhead": 1677.69, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA T4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 0.613363, "characteristics.90th_percentile_latency_ns": 613363.0, "characteristics.90th_percentile_latency_s": 0.000613363, "characteristics.90th_percentile_latency_us": 613.363, "characteristics.mAP": 22.914, "ck_system": "T4x1_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "ECC off", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/T4x1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 1024, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/T4x1_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1327.22, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "a4112b9d943d560e", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8379056, "90.00 percentile latency (ns)": 8485333, "90th percentile latency (ns)": 8485333, "95.00 percentile latency (ns)": 8553523, "97.00 percentile latency (ns)": 8598874, "99.00 percentile latency (ns)": 8646888, "99.90 percentile latency (ns)": 9209694, "Max latency (ns)": 19207371, "Mean latency (ns)": 8374879, "Min duration satisfied": "Yes", "Min latency (ns)": 6411941, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 118.34, "QPS w/o loadgen overhead": 119.4, "Result is": "VALID", "SUT name": "Triton_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA T4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 8.485333, "characteristics.90th_percentile_latency_ns": 8485333.0, "characteristics.90th_percentile_latency_s": 0.008485333, "characteristics.90th_percentile_latency_us": 8485.333, "characteristics.mAP": 20.111, "ck_system": "T4x1_TRT_Triton", "ck_used": true, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 120, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7V13 64-Core Processor", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "ECC off", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/T4x1_TRT_Triton", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.5 LTS (Linux-5.4.0-1055-azure-x86_64-with-Ubuntu-18.04-bionic)", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0, Triton 20.09; GCC 7.5.0; Python 3.7.10", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 64, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "Powered by CK v2.5.8 (https://github.com/ctuning/ck)", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/T4x1_TRT_Triton", "system_name": "Microsoft Corporation 7.0 (Virtual Machine)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 128.916, "task": "object detection", "task2": "object detection", "total_cores": 240, "uid": "2a63dc04395852e9", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 541365, "90.00 percentile latency (ns)": 550122, "90th percentile latency (ns)": 550122, "95.00 percentile latency (ns)": 553135, "97.00 percentile latency (ns)": 555247, "99.00 percentile latency (ns)": 560442, "99.90 percentile latency (ns)": 636447, "Max latency (ns)": 8471610, "Mean latency (ns)": 541776, "Min duration satisfied": "Yes", "Min latency (ns)": 506797, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 1815.75, "QPS w/o loadgen overhead": 1845.78, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA T4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 0.550122, "characteristics.90th_percentile_latency_ns": 550122.0, "characteristics.90th_percentile_latency_s": 0.000550122, "characteristics.90th_percentile_latency_us": 550.122, "characteristics.mAP": 22.914, "ck_system": "T4x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "ECC off", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/T4x1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 1024, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/T4x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x T4, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 1327.22, "task": "object detection", "task2": "object detection", "total_cores": 56, "uid": "15046da4f628fd9c", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 8260178, "90.00 percentile latency (ns)": 8416519, "90th percentile latency (ns)": 8416519, "95.00 percentile latency (ns)": 8455179, "97.00 percentile latency (ns)": 8471107, "99.00 percentile latency (ns)": 8513455, "99.90 percentile latency (ns)": 9129402, "Max latency (ns)": 12079917, "Mean latency (ns)": 8258500, "Min duration satisfied": "Yes", "Min latency (ns)": 6277321, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 120.0, "QPS w/o loadgen overhead": 121.09, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "16 GB", "accelerator_memory_configuration": "GDDR6", "accelerator_model_name": "NVIDIA T4", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 8.416519, "characteristics.90th_percentile_latency_ns": 8416519.0, "characteristics.90th_percentile_latency_s": 0.008416519, "characteristics.90th_percentile_latency_us": 8416.519, "characteristics.mAP": 20.111, "ck_system": "T4x1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 28, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8280 CPU @ 2.70GHz", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "ECC off", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/T4x1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 64, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/T4x1_TRT", "system_name": "Supermicro 4029GP-TRT-OTO-28 (1x T4, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 128.916, "task": "object detection", "task2": "object detection", "total_cores": 56, "uid": "f979492f8cec18eb", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 341074, "90.00 percentile latency (ns)": 350254, "90th percentile latency (ns)": 350254, "95.00 percentile latency (ns)": 354814, "97.00 percentile latency (ns)": 363824, "99.00 percentile latency (ns)": 376943, "99.90 percentile latency (ns)": 505471, "Max latency (ns)": 7238848, "Mean latency (ns)": 344149, "Min duration satisfied": "Yes", "Min latency (ns)": 305025, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 2804.16, "QPS w/o loadgen overhead": 2905.72, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 0.350254, "characteristics.90th_percentile_latency_ns": 350254.0, "characteristics.90th_percentile_latency_s": 0.000350254, "characteristics.90th_percentile_latency_us": 350.254, "characteristics.mAP": 22.914, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 1024, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z52 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 2173.91, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "2e9bff56aecd19a9", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1948128, "90.00 percentile latency (ns)": 1988396, "90th percentile latency (ns)": 1988396, "95.00 percentile latency (ns)": 1997896, "97.00 percentile latency (ns)": 2004855, "99.00 percentile latency (ns)": 2026026, "99.90 percentile latency (ns)": 2810113, "Max latency (ns)": 6275564, "Mean latency (ns)": 1950972, "Min duration satisfied": "Yes", "Min latency (ns)": 1849108, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 505.62, "QPS w/o loadgen overhead": 512.57, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "40GB", "accelerator_memory_configuration": "HBM2", "accelerator_model_name": "NVIDIA A100-PCIe", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 1.988396, "characteristics.90th_percentile_latency_ns": 1988396.0, "characteristics.90th_percentile_latency_s": 0.001988396, "characteristics.90th_percentile_latency_us": 1988.396, "characteristics.mAP": 20.111, "ck_system": "A100-PCIex1_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "TensorRT 7.2, CUDA 11.0 Update 1", "host_memory_capacity": "768 GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 64, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "AMD EPYC 7742", "host_processors_per_node": 2, "host_storage_capacity": "4 TB", "host_storage_type": "NVMe SSD", "hw_notes": "", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/A100-PCIex1_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "TensorRT 7.2, CUDA 11.0 Update 1, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 64, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/A100-PCIex1_TRT", "system_name": "Gigabyte G482-Z52 (1x A100-PCIe, TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 526.316, "task": "object detection", "task2": "object detection", "total_cores": 128, "uid": "4381f482f444289f", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 1652624, "90.00 percentile latency (ns)": 1674000, "90th percentile latency (ns)": 1674000, "95.00 percentile latency (ns)": 1683056, "97.00 percentile latency (ns)": 1690928, "99.00 percentile latency (ns)": 1718544, "99.90 percentile latency (ns)": 2704347, "Max latency (ns)": 24730769, "Mean latency (ns)": 1660474, "Min duration satisfied": "Yes", "Min latency (ns)": 1615440, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 599.41, "QPS w/o loadgen overhead": 602.24, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 1.674, "characteristics.90th_percentile_latency_ns": 1674000.0, "characteristics.90th_percentile_latency_s": 0.001674, "characteristics.90th_percentile_latency_us": 1674.0, "characteristics.mAP": 22.9, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2", "host_memory_capacity": "8GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-mobilenet", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 1024, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "ssd_mobilenet_v1_coco_2018_01_28/frozen_inference_graph.pb", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 308.452, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "cc400e9c8bf94228", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" }, { "50.00 percentile latency (ns)": 46752657, "90.00 percentile latency (ns)": 46907121, "90th percentile latency (ns)": 46907121, "95.00 percentile latency (ns)": 46951672, "97.00 percentile latency (ns)": 47049689, "99.00 percentile latency (ns)": 47318780, "99.90 percentile latency (ns)": 50199825, "Max latency (ns)": 54340367, "Mean latency (ns)": 46772874, "Min duration satisfied": "Yes", "Min latency (ns)": 46427893, "Min queries satisfied": "Yes", "Mode": "Performance", "QPS w/ loadgen overhead": 21.37, "QPS w/o loadgen overhead": 21.38, "Result is": "VALID", "SUT name": "LWIS_Server", "Scenario": "singlestream", "accelerator_frequency": "", "accelerator_host_interconnect": "", "accelerator_interconnect": "", "accelerator_interconnect_topology": "", "accelerator_memory_capacity": "Shared with host", "accelerator_memory_configuration": "SRAM", "accelerator_model_name": "NVIDIA Xavier NX", "accelerator_on-chip_memories": "", "accelerators_per_node": 1, "accuracy_log_probability": 0, "accuracy_log_rng_seed": 0, "accuracy_log_sampling_target": 0, "characteristics.90th_percentile_latency_ms": 46.907121, "characteristics.90th_percentile_latency_ns": 46907121.0, "characteristics.90th_percentile_latency_s": 0.046907121, "characteristics.90th_percentile_latency_us": 46907.121, "characteristics.mAP": 20.111, "ck_system": "Xavier_NX_TRT", "ck_used": false, "cooling": "", "dataset": "COCO 2017 (300x300)", "dataset_link": "https://github.com/ctuning/ck/blob/master/docs/mlperf-automation/datasets/coco2017.md", "dim_x_default": "characteristics.90th_percentile_latency_ms", "dim_y_default": "characteristics.mAP", "dim_y_maximize": true, "division": "closed", "formal_model": "ssd-mobilenet", "formal_model_accuracy": 99.0, "formal_model_link": "https://github.com/mlcommons/ck-mlops/tree/main/package", "framework": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2", "host_memory_capacity": "8GB", "host_memory_configuration": "", "host_networking": "", "host_networking_topology": "", "host_processor_caches": "", "host_processor_core_count": 6, "host_processor_frequency": "", "host_processor_interconnect": "", "host_processor_model_name": "NVIDIA Carmel (ARMv8.2)", "host_processors_per_node": 1, "host_storage_capacity": "32GB", "host_storage_type": "Micro SD Card", "hw_notes": "GPU and both DLAs are used in resnet50, ssd-mobilenet, and ssd-resnet34, in Offline and MultiStream scenarios", "informal_model": "ssd-resnet34", "input_data_types": "int8", "key.accuracy": "characteristics.mAP", "max_async_queries": 1, "max_duration (ms)": 0, "max_query_count": 0, "min_duration (ms)": 60000, "min_query_count": 1024, "mlperf_version": 0.7, "normalize_cores": 1, "normalize_processors": 1, "note_code": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/code", "note_details": "https://github.com/mlcommons/inference_results_v0.7/tree/master/closed/NVIDIA/results/Xavier_NX_TRT", "number_of_nodes": 1, "operating_system": "Ubuntu 18.04.4", "other_software_stack": "20.09 Jetson CUDA-X AI Developer Preview, TensorRT 7.2, CUDA 10.2, cuDNN 8.0.2, DALI 0.25.0", "performance_issue_same": true, "performance_issue_same_index": 0, "performance_issue_unique": true, "performance_sample_count": 64, "print_timestamps": true, "problem": false, "qsl_rng_seed": 12786827339337101903, "retraining": "N", "sample_index_rng_seed": 12640797754436136668, "samples_per_query": 1, "schedule_rng_seed": 3135815929913719677, "starting_weights_filename": "resnet34-ssd1200.pytorch", "status": "available", "submitter": "NVIDIA", "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/NVIDIA", "sw_notes": "", "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/Xavier_NX_TRT", "system_name": "NVIDIA Jetson Xavier NX (TensorRT)", "system_type": "edge", "target_latency (ns)": 0, "target_qps": 16.9618, "task": "object detection", "task2": "object detection", "total_cores": 6, "uid": "0cab71bb31c727da", "use_accelerator": true, "weight_data_types": "int8", "weight_transformations": "quantization, affine fusion" } ]