{"library":"perf-analyzer","type":"library","category":null,"description":"Triton Performance Analyzer (perf_analyzer) is a command-line interface (CLI) tool designed to optimize the inference performance of models running on the NVIDIA Triton Inference Server. It measures key metrics such as throughput and latency by generating inference requests to your model and repeating measurements until stable values are achieved. The library is currently at version 2.59.1 and follows the release cadence of the broader Triton Inference Server project.","language":"python","status":"active","version":"2.59.1","tags":["performance","benchmarking","inference","nvidia","triton","cli","latency","throughput"],"install":[{"cmd":"pip install perf-analyzer","imports":[]},{"cmd":"docker run --rm --gpus=all -it --net=host nvcr.io/nvidia/tritonserver:${RELEASE}-py3-sdk perf_analyzer -m <model>","imports":[]}],"homepage":"https://developer.nvidia.com/triton-inference-server","github":null,"docs":null,"changelog":null,"pypi":"https://pypi.org/project/perf-analyzer/","npm":null,"openapi_spec":null,"status_page":null,"smithery":null,"compatibility":null,"provenance":{"verified_status":null,"verified_at":null,"last_verified":"Mon Apr 13","next_check":"Sun Jul 12","install_tag":null}}