[
  {
    "50.00 percentile latency (ns)": 64291935,
    "90.00 percentile latency (ns)": 90945871,
    "95.00 percentile latency (ns)": 97833775,
    "97.00 percentile latency (ns)": 102715425,
    "99.00 percentile latency (ns)": 113476251,
    "99.90 percentile latency (ns)": 137036269,
    "Completed samples per second": 9985.02,
    "Max latency (ns)": 1024777779,
    "Mean latency (ns)": 65315136,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 18644940,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "Triton_MultiMigServer",
    "Scenario": "server",
    "Scheduled samples per second": 10001.94,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80 GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-PCIE-80GB (7x1g.10gb MIG)",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 10001.94,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 2500.485,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 2500.485,
    "ck_system": "R750xa_A100-PCIE-80GB-MIG_28x1g.10gb_TRT_Triton",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "open",
    "filesystem": "",
    "formal_model": "bert-99",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6338 CPU @ 2.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3.5 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/open/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/open/Dell/results/R750xa_A100-PCIE-80GB-MIG_28x1g.10gb_TRT_Triton",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R750xa_A100-PCIE-80GB-MIG_28x1g.10gb_TRT_Triton",
    "system_name": "Dell EMC PowerEdge R750xa (4x A100-PCIE-80GB-MIG-7x1g.10gb, TensorRT, Triton)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 10000,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "2642c5f288496abe",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 62691037,
    "90.00 percentile latency (ns)": 88709982,
    "95.00 percentile latency (ns)": 95049540,
    "97.00 percentile latency (ns)": 99252649,
    "99.00 percentile latency (ns)": 108463011,
    "99.90 percentile latency (ns)": 127541967,
    "Completed samples per second": 9985.15,
    "Max latency (ns)": 1025950343,
    "Mean latency (ns)": 63576614,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 19012047,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "Triton_MultiMigServer",
    "Scenario": "server",
    "Scheduled samples per second": 10001.94,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80 GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-PCIE-80GB (7x1g.10gb MIG)",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 10001.94,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 2500.485,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 2500.485,
    "ck_system": "R750xa_A100-PCIE-80GB-MIG_28x1g.10gb_TRT_Triton",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "open",
    "filesystem": "",
    "formal_model": "bert-99.9",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6338 CPU @ 2.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3.5 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/open/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/open/Dell/results/R750xa_A100-PCIE-80GB-MIG_28x1g.10gb_TRT_Triton",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R750xa_A100-PCIE-80GB-MIG_28x1g.10gb_TRT_Triton",
    "system_name": "Dell EMC PowerEdge R750xa (4x A100-PCIE-80GB-MIG-7x1g.10gb, TensorRT, Triton)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 10000,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "c8d841d49cea8bc7",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 71293531,
    "90.00 percentile latency (ns)": 92192885,
    "95.00 percentile latency (ns)": 97656178,
    "97.00 percentile latency (ns)": 101177155,
    "99.00 percentile latency (ns)": 107531497,
    "99.90 percentile latency (ns)": 117530109,
    "Completed samples per second": 7871.17,
    "Max latency (ns)": 132030117,
    "Mean latency (ns)": 71061853,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 3839961,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 7871.94,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "40 GB",
    "accelerator_memory_configuration": "HBM2",
    "accelerator_model_name": "NVIDIA A100-PCIE-40GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 3,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 7871.94,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 2623.98,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 2623.98,
    "ck_system": "R7525_A100-PCIE-40GBx3_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "512 GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 32,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "AMD EPYC 7502",
    "host_processors_per_node": 2,
    "host_storage_capacity": "1.8 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 3,
    "normalize_processors": 3,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R7525_A100-PCIE-40GBx3_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R7525_A100-PCIE-40GBx3_TRT",
    "system_name": "Dell EMC PowerEdge R7525 (3x A100-PCIE-40GB, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 7870,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 64,
    "uid": "55e6381234fe3bee",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 62630888,
    "90.00 percentile latency (ns)": 96210920,
    "95.00 percentile latency (ns)": 106761402,
    "97.00 percentile latency (ns)": 113787614,
    "99.00 percentile latency (ns)": 127185096,
    "99.90 percentile latency (ns)": 154211250,
    "Completed samples per second": 3814.6,
    "Max latency (ns)": 192997192,
    "Mean latency (ns)": 64973681,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 4134622,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 3814.94,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "40 GB",
    "accelerator_memory_configuration": "HBM2",
    "accelerator_model_name": "NVIDIA A100-PCIE-40GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 3,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 3814.94,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 1271.6466666666668,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 1271.6466666666668,
    "ck_system": "R7525_A100-PCIE-40GBx3_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "512 GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 32,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "AMD EPYC 7502",
    "host_processors_per_node": 2,
    "host_storage_capacity": "1.8 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 3,
    "normalize_processors": 3,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R7525_A100-PCIE-40GBx3_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R7525_A100-PCIE-40GBx3_TRT",
    "system_name": "Dell EMC PowerEdge R7525 (3x A100-PCIE-40GB, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 3812,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 64,
    "uid": "634a44bf249e180d",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 70316216,
    "90.00 percentile latency (ns)": 87321199,
    "95.00 percentile latency (ns)": 91314219,
    "97.00 percentile latency (ns)": 93687371,
    "99.00 percentile latency (ns)": 97935853,
    "99.90 percentile latency (ns)": 104795060,
    "Completed samples per second": 11700.14,
    "Max latency (ns)": 115825994,
    "Mean latency (ns)": 69636102,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 3765215,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 11701.09,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80 GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-PCIE-80GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 11701.09,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 2925.2725,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 2925.2725,
    "ck_system": "R750xa_A100-PCIE-80GBx4_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6338 CPU @ 2.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3.5 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R750xa_A100-PCIE-80GBx4_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R750xa_A100-PCIE-80GBx4_TRT",
    "system_name": "Dell EMC PowerEdge R750xa (4x A100-PCIE-80GB, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 11700,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "79ede446d59963a8",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 54406192,
    "90.00 percentile latency (ns)": 82743871,
    "95.00 percentile latency (ns)": 91616239,
    "97.00 percentile latency (ns)": 97562775,
    "99.00 percentile latency (ns)": 108990237,
    "99.90 percentile latency (ns)": 128668087,
    "Completed samples per second": 5682.42,
    "Max latency (ns)": 173148840,
    "Mean latency (ns)": 56261914,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 3955315,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 5682.9,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80 GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-PCIE-80GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 5682.9,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 1420.725,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 1420.725,
    "ck_system": "R750xa_A100-PCIE-80GBx4_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6338 CPU @ 2.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3.5 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R750xa_A100-PCIE-80GBx4_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R750xa_A100-PCIE-80GBx4_TRT",
    "system_name": "Dell EMC PowerEdge R750xa (4x A100-PCIE-80GB, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 5680,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "a29a53bb144d2584",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 52311358,
    "90.00 percentile latency (ns)": 69793972,
    "95.00 percentile latency (ns)": 75107740,
    "97.00 percentile latency (ns)": 78884576,
    "99.00 percentile latency (ns)": 89936109,
    "99.90 percentile latency (ns)": 525887628980,
    "Completed samples per second": 1849.26,
    "Max latency (ns)": 604725715053,
    "Mean latency (ns)": 2364501132,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 4628320,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 1849.4,
    "accelerator_cooling_type": "passive",
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24GB",
    "accelerator_memory_configuration": "GDDR6",
    "accelerator_model_name": "NVIDIA A10",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 2,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "2.12.3",
    "characteristics.power": 575.9031074380163,
    "characteristics.power.normalized_per_core": 287.95155371900813,
    "characteristics.power.normalized_per_processor": 287.95155371900813,
    "characteristics.scheduled_queries_per_second": 1849.4,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 924.7,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 924.7,
    "ck_system": "XE2420_A10x2_TRT_MaxQ",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "division": "closed",
    "filesystem": "ext3/ext4",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_cooling_type": "air",
    "host_memory_capacity": "384 GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 24,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6252 CPU @ 2.10GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "4 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "ECC on",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "5.00.00.00",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "normalize_cores": 2,
    "normalize_processors": 2,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/XE2420_A10x2_TRT_MaxQ",
    "number_of_nodes": 1,
    "operating_system": "Ubuntu 20.04.2",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.57.02, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_settings": "closed/Dell/power/XE2420_A10x2_power_settings.md",
    "print_timestamps": 0,
    "problem": false,
    "psu_details": "2x2000W",
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_cooling_type": "air",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/XE2420_A10x2_TRT_MaxQ",
    "system_name": "Dell EMC PowerEdge XE2420 (2x A10, MaxQ, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 1850,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 48,
    "uid": "91054c7b84a771fd",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 32811458,
    "90.00 percentile latency (ns)": 49135068,
    "95.00 percentile latency (ns)": 54498566,
    "97.00 percentile latency (ns)": 58401581,
    "99.00 percentile latency (ns)": 68363311,
    "99.90 percentile latency (ns)": 520053787894,
    "Completed samples per second": 839.88,
    "Max latency (ns)": 603882999992,
    "Mean latency (ns)": 2210618713,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 5114503,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 839.96,
    "accelerator_cooling_type": "passive",
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24GB",
    "accelerator_memory_configuration": "GDDR6",
    "accelerator_model_name": "NVIDIA A10",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 2,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "2.12.3",
    "characteristics.power": 583.2093884297519,
    "characteristics.power.normalized_per_core": 291.60469421487596,
    "characteristics.power.normalized_per_processor": 291.60469421487596,
    "characteristics.scheduled_queries_per_second": 839.96,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 419.98,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 419.98,
    "ck_system": "XE2420_A10x2_TRT_MaxQ",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "division": "closed",
    "filesystem": "ext3/ext4",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_cooling_type": "air",
    "host_memory_capacity": "384 GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 24,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6252 CPU @ 2.10GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "4 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "ECC on",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "5.00.00.00",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "normalize_cores": 2,
    "normalize_processors": 2,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/XE2420_A10x2_TRT_MaxQ",
    "number_of_nodes": 1,
    "operating_system": "Ubuntu 20.04.2",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.57.02, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_settings": "closed/Dell/power/XE2420_A10x2_power_settings.md",
    "print_timestamps": 0,
    "problem": false,
    "psu_details": "2x2000W",
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_cooling_type": "air",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/XE2420_A10x2_TRT_MaxQ",
    "system_name": "Dell EMC PowerEdge XE2420 (2x A10, MaxQ, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 840,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 48,
    "uid": "4c07b6d3136c8401",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 88508865,
    "90.00 percentile latency (ns)": 98210642,
    "95.00 percentile latency (ns)": 101702942,
    "97.00 percentile latency (ns)": 104197605,
    "99.00 percentile latency (ns)": 109636786,
    "99.90 percentile latency (ns)": 117904366,
    "Completed samples per second": 30101.76,
    "Max latency (ns)": 126362885,
    "Mean latency (ns)": 86625291,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 4273792,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 30105.21,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80 GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-PCIE-80GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 10,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 30105.21,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 3010.5209999999997,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 3010.5209999999997,
    "ck_system": "DSS8440_A100-PCIE-80GBx10_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": " 768GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 24,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6248R CPU @ 3.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "1.8 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 10,
    "normalize_processors": 10,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/DSS8440_A100-PCIE-80GBx10_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DSS8440_A100-PCIE-80GBx10_TRT",
    "system_name": "Dell EMC DSS 8440 (10x NVIDIA A100-PCIE-80GB, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 30110,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 48,
    "uid": "1e70489c8e7492d6",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 65952813,
    "90.00 percentile latency (ns)": 98075994,
    "95.00 percentile latency (ns)": 107620955,
    "97.00 percentile latency (ns)": 113828903,
    "99.00 percentile latency (ns)": 125979773,
    "99.90 percentile latency (ns)": 144188157,
    "Completed samples per second": 14537.95,
    "Max latency (ns)": 171586267,
    "Mean latency (ns)": 67329720,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 4183858,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 14539.86,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80 GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-PCIE-80GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 10,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 14539.86,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 1453.986,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 1453.986,
    "ck_system": "DSS8440_A100-PCIE-80GBx10_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": " 768GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 24,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6248R CPU @ 3.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "1.8 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 10,
    "normalize_processors": 10,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/DSS8440_A100-PCIE-80GBx10_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DSS8440_A100-PCIE-80GBx10_TRT",
    "system_name": "Dell EMC DSS 8440 (10x NVIDIA A100-PCIE-80GB, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 14540,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 48,
    "uid": "5f844cc8f45aed61",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 25272959,
    "90.00 percentile latency (ns)": 28157489,
    "95.00 percentile latency (ns)": 28921243,
    "97.00 percentile latency (ns)": 29423462,
    "99.00 percentile latency (ns)": 30430775,
    "99.90 percentile latency (ns)": 32596523,
    "Completed samples per second": 10684.43,
    "Max latency (ns)": 1003836133,
    "Mean latency (ns)": 25322524,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 17950123,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "Triton_Server",
    "Scenario": "server",
    "Scheduled samples per second": 10702.3,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80 GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-PCIE-80GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 10702.3,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 2675.575,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 2675.575,
    "ck_system": "R750xa_A100-PCIE-80GBx4_TRT_Triton",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6338 CPU @ 2.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3.5 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R750xa_A100-PCIE-80GBx4_TRT_Triton",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R750xa_A100-PCIE-80GBx4_TRT_Triton",
    "system_name": "Dell EMC PowerEdge R750xa (4x A100-PCIE-80GB, TensorRT, Triton)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 10700,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "ed7c317449c4509a",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 51936385,
    "90.00 percentile latency (ns)": 61086474,
    "95.00 percentile latency (ns)": 64736121,
    "97.00 percentile latency (ns)": 67444395,
    "99.00 percentile latency (ns)": 73445590,
    "99.90 percentile latency (ns)": 85634473,
    "Completed samples per second": 5673.36,
    "Max latency (ns)": 1010193203,
    "Mean latency (ns)": 52757927,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 35540559,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "Triton_Server",
    "Scenario": "server",
    "Scheduled samples per second": 5682.9,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80 GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-PCIE-80GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 5682.9,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 1420.725,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 1420.725,
    "ck_system": "R750xa_A100-PCIE-80GBx4_TRT_Triton",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6338 CPU @ 2.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3.5 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R750xa_A100-PCIE-80GBx4_TRT_Triton",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R750xa_A100-PCIE-80GBx4_TRT_Triton",
    "system_name": "Dell EMC PowerEdge R750xa (4x A100-PCIE-80GB, TensorRT, Triton)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 5680,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "04d7dde6e4d54bbe",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 47421110,
    "90.00 percentile latency (ns)": 70437645,
    "95.00 percentile latency (ns)": 77905060,
    "97.00 percentile latency (ns)": 83191800,
    "99.00 percentile latency (ns)": 98228209,
    "99.90 percentile latency (ns)": 529143559242,
    "Completed samples per second": 11500.39,
    "Max latency (ns)": 604119503762,
    "Mean latency (ns)": 2279000376,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 4602312,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 11501.43,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24 GB",
    "accelerator_memory_configuration": "HBM2",
    "accelerator_model_name": "NVIDIA A30",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 8,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 11501.43,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 1437.67875,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 1437.67875,
    "ck_system": "DSS8440_A30x8_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 48,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6248R CPU @ 3.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3 TB",
    "host_storage_type": "NVMe",
    "hw_notes": "",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 8,
    "normalize_processors": 8,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/DSS8440_A30x8_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DSS8440_A30x8_TRT",
    "system_name": "Dell EMC DSS 8440 (8x A30, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 11500,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 96,
    "uid": "4d42b64b64fe2b70",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 45564485,
    "90.00 percentile latency (ns)": 72017673,
    "95.00 percentile latency (ns)": 81039942,
    "97.00 percentile latency (ns)": 87530118,
    "99.00 percentile latency (ns)": 107117729,
    "99.90 percentile latency (ns)": 518496846984,
    "Completed samples per second": 5252.4,
    "Max latency (ns)": 603709061971,
    "Mean latency (ns)": 2231764767,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 5478794,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 5252.86,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24 GB",
    "accelerator_memory_configuration": "HBM2",
    "accelerator_model_name": "NVIDIA A30",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 8,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 5252.86,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 656.6075,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 656.6075,
    "ck_system": "DSS8440_A30x8_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 48,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6248R CPU @ 3.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3 TB",
    "host_storage_type": "NVMe",
    "hw_notes": "",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 8,
    "normalize_processors": 8,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/DSS8440_A30x8_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DSS8440_A30x8_TRT",
    "system_name": "Dell EMC DSS 8440 (8x A30, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 5250,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 96,
    "uid": "877e33499fbb997e",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 33601498,
    "90.00 percentile latency (ns)": 47176338,
    "95.00 percentile latency (ns)": 51580488,
    "97.00 percentile latency (ns)": 54759352,
    "99.00 percentile latency (ns)": 65008940,
    "99.90 percentile latency (ns)": 523740978212,
    "Completed samples per second": 3962.03,
    "Max latency (ns)": 604144536883,
    "Mean latency (ns)": 2523178139,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 4980738,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 3962.46,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24 GB",
    "accelerator_memory_configuration": "HBM2",
    "accelerator_model_name": "NVIDIA A30",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 3,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 3962.46,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 1320.82,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 1320.82,
    "ck_system": "R7525_A30x3_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "AMD EPYC 7763",
    "host_processors_per_node": 2,
    "host_storage_capacity": "4 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 3,
    "normalize_processors": 3,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R7525_A30x3_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.4.2105",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R7525_A30x3_TRT",
    "system_name": "Dell EMC PowerEdge R7525 (3x A30, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 3960,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "602061a04ebf921b",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 47150053,
    "90.00 percentile latency (ns)": 73212610,
    "95.00 percentile latency (ns)": 81397863,
    "97.00 percentile latency (ns)": 87347928,
    "99.00 percentile latency (ns)": 108768653,
    "99.90 percentile latency (ns)": 528708960251,
    "Completed samples per second": 1949.8,
    "Max latency (ns)": 604021013265,
    "Mean latency (ns)": 2555961939,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 5747642,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 1949.88,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24 GB",
    "accelerator_memory_configuration": "HBM2",
    "accelerator_model_name": "NVIDIA A30",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 3,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 1949.88,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 649.96,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 649.96,
    "ck_system": "R7525_A30x3_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "AMD EPYC 7763",
    "host_processors_per_node": 2,
    "host_storage_capacity": "4 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 3,
    "normalize_processors": 3,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R7525_A30x3_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.4.2105",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R7525_A30x3_TRT",
    "system_name": "Dell EMC PowerEdge R7525 (3x A30, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 1950,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "9b90b57614eb48aa",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 66823035,
    "90.00 percentile latency (ns)": 86015858,
    "95.00 percentile latency (ns)": 90986540,
    "97.00 percentile latency (ns)": 94118867,
    "99.00 percentile latency (ns)": 99737500,
    "99.90 percentile latency (ns)": 108636542,
    "Completed samples per second": 7800.84,
    "Max latency (ns)": 122959662,
    "Mean latency (ns)": 66525410,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 8357397,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 7801.53,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "40 GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-PCIE-40GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 3,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 7801.53,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 2600.5099999999998,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 2600.5099999999998,
    "ck_system": "R7525_vA100-PCIE-40GBx3_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "AMD EPYC 7502",
    "host_processors_per_node": 2,
    "host_storage_capacity": "1.8 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 3,
    "normalize_processors": 3,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R7525_vA100-PCIE-40GBx3_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "VMware Submission",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R7525_vA100-PCIE-40GBx3_TRT",
    "system_name": "Dell EMC PowerEdge R7525 (3x GRID A100-40C, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 7800,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "1f0459015a04a839",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 39735573,
    "90.00 percentile latency (ns)": 60423352,
    "95.00 percentile latency (ns)": 67279242,
    "97.00 percentile latency (ns)": 72087978,
    "99.00 percentile latency (ns)": 82080334,
    "99.90 percentile latency (ns)": 100399717,
    "Completed samples per second": 3601.91,
    "Max latency (ns)": 124926508,
    "Mean latency (ns)": 41530372,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 3713171,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 3602.11,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "40 GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-PCIE-40GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 3,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 3602.11,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 1200.7033333333334,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 1200.7033333333334,
    "ck_system": "R7525_vA100-PCIE-40GBx3_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "AMD EPYC 7502",
    "host_processors_per_node": 2,
    "host_storage_capacity": "1.8 TB",
    "host_storage_type": "SSD",
    "hw_notes": "",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 3,
    "normalize_processors": 3,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R7525_vA100-PCIE-40GBx3_TRT",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "VMware Submission",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R7525_vA100-PCIE-40GBx3_TRT",
    "system_name": "Dell EMC PowerEdge R7525 (3x GRID A100-40C, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 3600,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "389049f3874283e5",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 63389773,
    "90.00 percentile latency (ns)": 95412837,
    "95.00 percentile latency (ns)": 104637541,
    "97.00 percentile latency (ns)": 110495125,
    "99.00 percentile latency (ns)": 121160556,
    "99.90 percentile latency (ns)": 137457581,
    "Completed samples per second": 7800.89,
    "Max latency (ns)": 153931697,
    "Mean latency (ns)": 65390188,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 7063083,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 7801.53,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "40 GB",
    "accelerator_memory_configuration": "HBM2",
    "accelerator_model_name": "NVIDIA A100-PCIE-40GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.power": 1480.3514143094853,
    "characteristics.power.normalized_per_core": 370.08785357737133,
    "characteristics.power.normalized_per_processor": 370.08785357737133,
    "characteristics.scheduled_queries_per_second": 7801.53,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 1950.3825,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 1950.3825,
    "ck_system": "R750xa_A100-PCIE-40GBx4_TRT_MaxQ",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "256 GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 32,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8368 CPU @ 2.40GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "543 GB",
    "host_storage_type": "SSD",
    "hw_notes": "Result Measured for Power; GPU Power Limited to 175W",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R750xa_A100-PCIE-40GBx4_TRT_MaxQ",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_settings": "closed/Dell/power/R750xa_A100-PCIE-40GBx4_TRT_MaxQ.md",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "1x2400W",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R750xa_A100-PCIE-40GBx4_TRT_MaxQ",
    "system_name": "Dell EMC PowerEdge R750xa (4x A100-PCIE-40GB, MaxQ, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 7800,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 64,
    "uid": "9484c42a2102a661",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 38969342,
    "90.00 percentile latency (ns)": 66029902,
    "95.00 percentile latency (ns)": 76617775,
    "97.00 percentile latency (ns)": 84188817,
    "99.00 percentile latency (ns)": 101087927,
    "99.90 percentile latency (ns)": 136299538,
    "Completed samples per second": 3551.57,
    "Max latency (ns)": 184708896,
    "Mean latency (ns)": 42487815,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 4457785,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 3551.72,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "40 GB",
    "accelerator_memory_configuration": "HBM2",
    "accelerator_model_name": "NVIDIA A100-PCIE-40GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.power": 1468.5383333333336,
    "characteristics.power.normalized_per_core": 367.1345833333334,
    "characteristics.power.normalized_per_processor": 367.1345833333334,
    "characteristics.scheduled_queries_per_second": 3551.72,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 887.93,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 887.93,
    "ck_system": "R750xa_A100-PCIE-40GBx4_TRT_MaxQ",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "256 GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 32,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Platinum 8368 CPU @ 2.40GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "543 GB",
    "host_storage_type": "SSD",
    "hw_notes": "Result Measured for Power; GPU Power Limited to 175W",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/R750xa_A100-PCIE-40GBx4_TRT_MaxQ",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_settings": "closed/Dell/power/R750xa_A100-PCIE-40GBx4_TRT_MaxQ.md",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "1x2400W",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/R750xa_A100-PCIE-40GBx4_TRT_MaxQ",
    "system_name": "Dell EMC PowerEdge R750xa (4x A100-PCIE-40GB, MaxQ, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 3550,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 64,
    "uid": "93fd738f7b40218d",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 59898136,
    "90.00 percentile latency (ns)": 79347045,
    "95.00 percentile latency (ns)": 86081095,
    "97.00 percentile latency (ns)": 91424326,
    "99.00 percentile latency (ns)": 107472592,
    "99.90 percentile latency (ns)": 526429959177,
    "Completed samples per second": 1949.7,
    "Max latency (ns)": 604678835154,
    "Mean latency (ns)": 2299372247,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 4648634,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 1949.88,
    "accelerator_cooling_type": "passive",
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24GB",
    "accelerator_memory_configuration": "GDDR6",
    "accelerator_model_name": "NVIDIA A10",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 2,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "2.12.3",
    "characteristics.scheduled_queries_per_second": 1949.88,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 974.94,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 974.94,
    "ck_system": "XE2420_A10x2_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "division": "closed",
    "filesystem": "ext3/ext4",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_cooling_type": "air",
    "host_memory_capacity": "384 GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 24,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6252 CPU @ 2.10GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "4 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "ECC on",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "5.00.00.00",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "normalize_cores": 2,
    "normalize_processors": 2,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/XE2420_A10x2_TRT",
    "number_of_nodes": 1,
    "operating_system": "Ubuntu 20.04.2",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.57.02, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_settings": "closed/Dell/power/XE2420_A10x2_power_settings.md",
    "print_timestamps": 0,
    "problem": false,
    "psu_details": "2x2000W",
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_cooling_type": "air",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/XE2420_A10x2_TRT",
    "system_name": "Dell EMC PowerEdge XE2420 (2x A10, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 1950,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 48,
    "uid": "cae894e1586ec4e5",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 49392033,
    "90.00 percentile latency (ns)": 71982132,
    "95.00 percentile latency (ns)": 79018988,
    "97.00 percentile latency (ns)": 84045360,
    "99.00 percentile latency (ns)": 98448322,
    "99.90 percentile latency (ns)": 523274890590,
    "Completed samples per second": 939.65,
    "Max latency (ns)": 604507345385,
    "Mean latency (ns)": 2286692869,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 5185766,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 939.74,
    "accelerator_cooling_type": "passive",
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24GB",
    "accelerator_memory_configuration": "GDDR6",
    "accelerator_model_name": "NVIDIA A10",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 2,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "2.12.3",
    "characteristics.scheduled_queries_per_second": 939.74,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 469.87,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 469.87,
    "ck_system": "XE2420_A10x2_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "division": "closed",
    "filesystem": "ext3/ext4",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_cooling_type": "air",
    "host_memory_capacity": "384 GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 24,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6252 CPU @ 2.10GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "4 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "ECC on",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "5.00.00.00",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "normalize_cores": 2,
    "normalize_processors": 2,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/XE2420_A10x2_TRT",
    "number_of_nodes": 1,
    "operating_system": "Ubuntu 20.04.2",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.57.02, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_settings": "closed/Dell/power/XE2420_A10x2_power_settings.md",
    "print_timestamps": 0,
    "problem": false,
    "psu_details": "2x2000W",
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_cooling_type": "air",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/XE2420_A10x2_TRT",
    "system_name": "Dell EMC PowerEdge XE2420 (2x A10, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 940,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 48,
    "uid": "49718fe12b38453b",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 25085622,
    "90.00 percentile latency (ns)": 44274974,
    "95.00 percentile latency (ns)": 46301902,
    "97.00 percentile latency (ns)": 47385041,
    "99.00 percentile latency (ns)": 49218454,
    "99.90 percentile latency (ns)": 52093753,
    "Completed samples per second": 11001.14,
    "Max latency (ns)": 56783405,
    "Mean latency (ns)": 26393681,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 4875441,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "Triton_Server",
    "Scenario": "server",
    "Scheduled samples per second": 11001.55,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24 GB",
    "accelerator_memory_configuration": "HBM2",
    "accelerator_model_name": "NVIDIA A30",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 8,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 11001.55,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 1375.19375,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 1375.19375,
    "ck_system": "DSS8440_A30x8_TRT_Triton",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 48,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6248R CPU @ 3.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3 TB",
    "host_storage_type": "NVMe",
    "hw_notes": "",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 8,
    "normalize_processors": 8,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/DSS8440_A30x8_TRT_Triton",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DSS8440_A30x8_TRT_Triton",
    "system_name": "Dell EMC DSS 8440 (8x A30, TensorRT, Triton)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 11000,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 96,
    "uid": "a107d6a41ebe4d7b",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 91191708,
    "90.00 percentile latency (ns)": 98320950,
    "95.00 percentile latency (ns)": 100248661,
    "97.00 percentile latency (ns)": 101509438,
    "99.00 percentile latency (ns)": 104021958,
    "99.90 percentile latency (ns)": 108401770,
    "Completed samples per second": 5193.88,
    "Max latency (ns)": 1054171786,
    "Mean latency (ns)": 91230949,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 69342737,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "Triton_Server",
    "Scenario": "server",
    "Scheduled samples per second": 5202.95,
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24 GB",
    "accelerator_memory_configuration": "HBM2",
    "accelerator_model_name": "NVIDIA A30",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 8,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 5202.95,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 650.36875,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 650.36875,
    "ck_system": "DSS8440_A30x8_TRT_Triton",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "disk_controllers": "",
    "disk_drives": "",
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 48,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6248R CPU @ 3.00GHz",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3 TB",
    "host_storage_type": "NVMe",
    "hw_notes": "",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "network_speed_mbit": "",
    "nics_enabled_connected": "",
    "nics_enabled_firmware": "",
    "nics_enabled_os": "",
    "normalize_cores": 8,
    "normalize_processors": 8,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/DSS8440_A30x8_TRT_Triton",
    "number_of_nodes": 1,
    "number_of_type_nics_installed": "",
    "operating_system": "CentOS 8.2",
    "other_hardware": "",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.42.01, DALI 0.31.0, Triton 21.07",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_supply_details": "",
    "power_supply_quantity_and_rating_watts": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/DSS8440_A30x8_TRT_Triton",
    "system_name": "Dell EMC DSS 8440 (8x A30, TensorRT, Triton)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 5200,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 96,
    "uid": "694139d83d490cc3",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 51096716,
    "90.00 percentile latency (ns)": 67924475,
    "95.00 percentile latency (ns)": 72895741,
    "97.00 percentile latency (ns)": 76263583,
    "99.00 percentile latency (ns)": 85785994,
    "99.90 percentile latency (ns)": 523597781104,
    "Completed samples per second": 1849.3,
    "Max latency (ns)": 604365920544,
    "Mean latency (ns)": 2245331913,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 4505543,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 1849.4,
    "accelerator_cooling_type": "",
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24GB",
    "accelerator_memory_configuration": "GDDR6",
    "accelerator_model_name": "NVIDIA A10",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 2,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 1849.4,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 924.7,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 924.7,
    "ck_system": "XR12_datacenter_A10x2_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_cooling_type": "",
    "host_memory_capacity": "512 GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 28,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6330 CPU @ 2.00GHz",
    "host_processors_per_node": 1,
    "host_storage_capacity": "4 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "ECC on",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "normalize_cores": 2,
    "normalize_processors": 2,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/XR12_datacenter_A10x2_TRT",
    "number_of_nodes": 1,
    "operating_system": "Ubuntu 20.04.2",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.57.02, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_settings": "",
    "print_timestamps": 0,
    "problem": false,
    "psu_details": "",
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_cooling_type": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/XR12_datacenter_A10x2_TRT",
    "system_name": "Dell EMC PowerEdge XR12 (2x A10, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 1850,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 28,
    "uid": "171b37238a499bba",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 32256124,
    "90.00 percentile latency (ns)": 48066811,
    "95.00 percentile latency (ns)": 53160450,
    "97.00 percentile latency (ns)": 56854434,
    "99.00 percentile latency (ns)": 66583382,
    "99.90 percentile latency (ns)": 514777335830,
    "Completed samples per second": 839.89,
    "Max latency (ns)": 603343368170,
    "Mean latency (ns)": 2126932883,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 5187928,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 839.96,
    "accelerator_cooling_type": "",
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "24GB",
    "accelerator_memory_configuration": "GDDR6",
    "accelerator_model_name": "NVIDIA A10",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 2,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "",
    "characteristics.scheduled_queries_per_second": 839.96,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 419.98,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 419.98,
    "ck_system": "XR12_datacenter_A10x2_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "division": "closed",
    "filesystem": "",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_cooling_type": "",
    "host_memory_capacity": "512 GB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 28,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "Intel(R) Xeon(R) Gold 6330 CPU @ 2.00GHz",
    "host_processors_per_node": 1,
    "host_storage_capacity": "4 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "ECC on",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "normalize_cores": 2,
    "normalize_processors": 2,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/XR12_datacenter_A10x2_TRT",
    "number_of_nodes": 1,
    "operating_system": "Ubuntu 20.04.2",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.57.02, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_settings": "",
    "print_timestamps": 0,
    "problem": false,
    "psu_details": "",
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_cooling_type": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/XR12_datacenter_A10x2_TRT",
    "system_name": "Dell EMC PowerEdge XR12 (2x A10, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 840,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 28,
    "uid": "5a344b7581715887",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 37890201,
    "90.00 percentile latency (ns)": 47348304,
    "95.00 percentile latency (ns)": 50935886,
    "97.00 percentile latency (ns)": 53415004,
    "99.00 percentile latency (ns)": 58099511,
    "99.90 percentile latency (ns)": 65088037,
    "Completed samples per second": 12905.98,
    "Max latency (ns)": 1014045620,
    "Mean latency (ns)": 39100689,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 26100422,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "Triton_Server",
    "Scenario": "server",
    "Scheduled samples per second": 12927.72,
    "accelerator_cooling_type": "",
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-SXM-80GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "characteristics.scheduled_queries_per_second": 12927.72,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 3231.93,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 3231.93,
    "ck_system": "XE8545_A100-SXM-80GBx4_TRT_Triton",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "division": "closed",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_cooling_type": "",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "AMD EPYC 7763",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "500W A100-SXM-80GB",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/XE8545_A100-SXM-80GBx4_TRT_Triton",
    "number_of_nodes": 1,
    "operating_system": "Ubuntu 20.04.2",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.57.02, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/XE8545_A100-SXM-80GBx4_TRT_Triton",
    "system_name": "Dell EMC PowerEdge XE8545 (4x A100-SXM-80GB, TensorRT, Triton)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 12926,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "d6545c0bc37a90d5",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 42430248,
    "90.00 percentile latency (ns)": 49114940,
    "95.00 percentile latency (ns)": 51756842,
    "97.00 percentile latency (ns)": 53688357,
    "99.00 percentile latency (ns)": 57946051,
    "99.90 percentile latency (ns)": 67958067,
    "Completed samples per second": 6390.81,
    "Max latency (ns)": 1034180078,
    "Mean latency (ns)": 42935280,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 28769317,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "Triton_Server",
    "Scenario": "server",
    "Scheduled samples per second": 6401.69,
    "accelerator_cooling_type": "",
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-SXM-80GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "characteristics.scheduled_queries_per_second": 6401.69,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 1600.4225,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 1600.4225,
    "ck_system": "XE8545_A100-SXM-80GBx4_TRT_Triton",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "division": "closed",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_cooling_type": "",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "AMD EPYC 7763",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "500W A100-SXM-80GB",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/XE8545_A100-SXM-80GBx4_TRT_Triton",
    "number_of_nodes": 1,
    "operating_system": "Ubuntu 20.04.2",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.57.02, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "print_timestamps": 0,
    "problem": false,
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/XE8545_A100-SXM-80GBx4_TRT_Triton",
    "system_name": "Dell EMC PowerEdge XE8545 (4x A100-SXM-80GB, TensorRT, Triton)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 6400,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "8e01a5ecff4a989c",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 58619850,
    "90.00 percentile latency (ns)": 67084445,
    "95.00 percentile latency (ns)": 70162212,
    "97.00 percentile latency (ns)": 72629582,
    "99.00 percentile latency (ns)": 89091485,
    "99.90 percentile latency (ns)": 545916104106,
    "Completed samples per second": 13790.63,
    "Max latency (ns)": 606525496737,
    "Mean latency (ns)": 3074482424,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 3836578,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 13791.67,
    "accelerator_cooling_type": "",
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-SXM-80GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "2.2.5",
    "characteristics.power": 3245.7612850082396,
    "characteristics.power.normalized_per_core": 811.4403212520599,
    "characteristics.power.normalized_per_processor": 811.4403212520599,
    "characteristics.scheduled_queries_per_second": 13791.67,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 3447.9175,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 3447.9175,
    "ck_system": "XE8545_A100-SXM-80GBx4_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "division": "closed",
    "filesystem": "ext3/ext4",
    "formal_model": "bert",
    "formal_model_accuracy": 99.0,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_cooling_type": "",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "AMD EPYC 7763",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "500W A100-SXM-80GB",
    "informal_model": "bert-99",
    "input_data_types": "int32",
    "management_firmware_version": "4.40.40.151",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/XE8545_A100-SXM-80GBx4_TRT",
    "number_of_nodes": 1,
    "operating_system": "Ubuntu 20.04.2",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.57.02, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_settings": "closed/Dell/power/XE8545_A100-SXM-80GBx4_power_settings.md",
    "print_timestamps": 0,
    "problem": false,
    "psu_details": "4x2400W",
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_cooling_type": "air",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/XE8545_A100-SXM-80GBx4_TRT",
    "system_name": "Dell EMC PowerEdge XE8545 (4x A100-SXM-80GB, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 13790,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "220a65d6cd83acb7",
    "use_accelerator": true,
    "weight_data_types": "int8",
    "weight_transformations": "quantization, affine fusion"
  },
  {
    "50.00 percentile latency (ns)": 47449275,
    "90.00 percentile latency (ns)": 58651482,
    "95.00 percentile latency (ns)": 61405258,
    "97.00 percentile latency (ns)": 63430451,
    "99.00 percentile latency (ns)": 75175733,
    "99.90 percentile latency (ns)": 544381073819,
    "Completed samples per second": 6901.69,
    "Max latency (ns)": 606801462819,
    "Mean latency (ns)": 3057311248,
    "Min duration satisfied": "Yes",
    "Min latency (ns)": 4363421,
    "Min queries satisfied": "Yes",
    "Mode": "PerformanceOnly",
    "Performance constraints satisfied": "Yes",
    "Result is": "VALID",
    "SUT name": "BERT SERVER",
    "Scenario": "server",
    "Scheduled samples per second": 6902.14,
    "accelerator_cooling_type": "",
    "accelerator_frequency": "",
    "accelerator_host_interconnect": "",
    "accelerator_interconnect": "",
    "accelerator_interconnect_topology": "",
    "accelerator_memory_capacity": "80GB",
    "accelerator_memory_configuration": "HBM2e",
    "accelerator_model_name": "NVIDIA A100-SXM-80GB",
    "accelerator_on-chip_memories": "",
    "accelerators_per_node": 4,
    "accuracy_log_probability": 0,
    "accuracy_log_rng_seed": 0,
    "accuracy_log_sampling_target": 0,
    "boot_firmware_version": "2.2.5",
    "characteristics.power": 3139.3082372322874,
    "characteristics.power.normalized_per_core": 784.8270593080719,
    "characteristics.power.normalized_per_processor": 784.8270593080719,
    "characteristics.scheduled_queries_per_second": 6902.14,
    "characteristics.scheduled_queries_per_second.normalized_per_core": 1725.535,
    "characteristics.scheduled_queries_per_second.normalized_per_processor": 1725.535,
    "ck_system": "XE8545_A100-SXM-80GBx4_TRT",
    "ck_used": false,
    "cooling": "",
    "dataset": "SQuAD v1.1",
    "dataset_link": "",
    "dim_x_default": "seq_number",
    "dim_y_default": "characteristics.scheduled_queries_per_second",
    "dim_y_maximize": false,
    "division": "closed",
    "filesystem": "ext3/ext4",
    "formal_model": "bert",
    "formal_model_accuracy": 99.9,
    "formal_model_link": "",
    "framework": "TensorRT 8.0.2, CUDA 11.3",
    "host_cooling_type": "",
    "host_memory_capacity": "1 TB",
    "host_memory_configuration": "",
    "host_networking": "",
    "host_networking_topology": "",
    "host_processor_caches": "",
    "host_processor_core_count": 64,
    "host_processor_frequency": "",
    "host_processor_interconnect": "",
    "host_processor_model_name": "AMD EPYC 7763",
    "host_processors_per_node": 2,
    "host_storage_capacity": "3 TB",
    "host_storage_type": "NVMe SSD",
    "hw_notes": "500W A100-SXM-80GB",
    "informal_model": "bert-99.9",
    "input_data_types": "int32",
    "management_firmware_version": "4.40.40.151",
    "max_async_queries": 0,
    "max_duration (ms)": 0,
    "max_query_count": 0,
    "min_duration (ms)": 600000,
    "min_query_count": 270336,
    "mlperf_version": 1.1,
    "normalize_cores": 4,
    "normalize_processors": 4,
    "note_code": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/code",
    "note_details": "https://github.com/mlcommons/inference_results_v1.1/tree/master/closed/Dell/results/XE8545_A100-SXM-80GBx4_TRT",
    "number_of_nodes": 1,
    "operating_system": "Ubuntu 20.04.2",
    "other_software_stack": "TensorRT 8.0.2, CUDA 11.3, cuDNN 8.2.1, Driver 470.57.02, DALI 0.31.0",
    "performance_issue_same": 0,
    "performance_issue_same_index": 0,
    "performance_issue_unique": 0,
    "performance_sample_count": 10833,
    "power_management": "",
    "power_settings": "closed/Dell/power/XE8545_A100-SXM-80GBx4_power_settings.md",
    "print_timestamps": 0,
    "problem": false,
    "psu_details": "4x2400W",
    "qsl_rng_seed": 1624344308455410291,
    "retraining": "N",
    "sample_index_rng_seed": 517984244576520566,
    "samples_per_query": 1,
    "schedule_rng_seed": 10051496985653635065,
    "starting_weights_filename": "bert_large_v1_1_fake_quant.onnx",
    "status": "available",
    "submitter": "Dell",
    "submitter_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.submitter/Dell",
    "sw_notes": "",
    "system_cooling_type": "air",
    "system_link": "https://github.com/ctuning/ck-mlperf-inference/tree/main/bench.mlperf.system/XE8545_A100-SXM-80GBx4_TRT",
    "system_name": "Dell EMC PowerEdge XE8545 (4x A100-SXM-80GB, TensorRT)",
    "system_type": "datacenter",
    "target_latency (ns)": 130000000,
    "target_qps": 6900,
    "task": "NLP",
    "task2": "nlp",
    "total_cores": 128,
    "uid": "3b504f2487db81e0",
    "use_accelerator": true,
    "weight_data_types": "fp16",
    "weight_transformations": "quantization, affine fusion"
  }
]