{
  "model": "Orpheus-3b-FT-Q4_K_M.gguf",
  "torch": "2.12.0+cpu",
  "snac_device": "cpu",
  "snac_threads": 4,
  "text": "Hey Aku, the MacBook is handling my voice now. Your gaming GPU can take the night off.",
  "runs": [
    {
      "run": 1,
      "elapsed_seconds": 27.717132244000823,
      "audio_seconds": 6.912,
      "realtime_factor": 4.010001771412156,
      "audio_tokens": 588,
      "first_token_seconds": 0.2219807130004483,
      "first_audio_seconds": 1.1999989340001775,
      "generation_stream_seconds": 27.716824506000194,
      "decoder_busy_seconds": 5.303662492991862,
      "timings": {
        "cache_n": 0,
        "prompt_n": 27,
        "prompt_ms": 219.873,
        "prompt_per_token_ms": 8.143444444444444,
        "prompt_per_second": 122.79816075643667,
        "predicted_n": 617,
        "predicted_ms": 27494.889,
        "predicted_per_token_ms": 44.562218800648296,
        "predicted_per_second": 22.44053431166789
      },
      "stopped_limit": null,
      "swap": "total        used        free      shared  buff/cache   available\nMem:           31540       22671        1169        8195       16349        8869\nSwap:              0           0           0"
    }
  ],
  "folded_weight_count": 71,
  "torch_config": "PyTorch built with:\n  - GCC 13.3\n  - C++ Version: 202002\n  - Intel(R) oneAPI Math Kernel Library Version 2024.2-Product Build 20240605 for Intel(R) 64 architecture applications\n  - Intel(R) MKL-DNN v3.11.2 (Git Hash 03c022d3ffdcee958cfacbe720048e725fdf644c)\n  - OpenMP 201511 (a.k.a. OpenMP 4.5)\n  - LAPACK is enabled (usually provided by MKL)\n  - NNPACK is enabled\n  - CPU capability usage: AVX2\n  - Build settings: BLAS_INFO=mkl, BUILD_TYPE=Release, COMMIT_SHA=7661cd9c6b841b62b7f411aa52ec51f05457263b, CUDA_FLAGS= -DLIBCUDACXX_ENABLE_SIMPLIFIED_COMPLEX_OPERATIONS -Xfatbin -compress-all  -Wno-deprecated-gpu-targets --expt-extended-lambda -DCUB_WRAPPED_NAMESPACE=at_cuda_detail -DDISABLE_CUSPARSE_DEPRECATED -DCUDA_HAS_FP16=1 -D__CUDA_NO_HALF_OPERATORS__ -D__CUDA_NO_HALF_CONVERSIONS__ -D__CUDA_NO_HALF2_OPERATORS__ -D__CUDA_NO_BFLOAT16_CONVERSIONS__ -DC10_NODEPRECATED, CXX_COMPILER=/opt/rh/gcc-toolset-13/root/usr/bin/c++, CXX_FLAGS= -fvisibility-inlines-hidden -DUSE_PTHREADPOOL -DNDEBUG -DUSE_KINETO -DLIBKINETO_NOCUPTI -DLIBKINETO_NOROCTRACER -DLIBKINETO_NOXPUPTI=ON -DUSE_FBGEMM -DUSE_PYTORCH_QNNPACK -DUSE_XNNPACK -DSYMBOLICATE_MOBILE_DEBUG_HANDLE -O2 -fPIC -DC10_NODEPRECATED -Wall -Wextra -Werror=return-type -Werror=non-virtual-dtor -Werror=range-loop-construct -Werror=bool-operation -Wnarrowing -Wno-missing-field-initializers -Wno-unknown-pragmas -Wno-unused-parameter -Wno-strict-overflow -Wno-strict-aliasing -Wno-stringop-overflow -Wsuggest-override -Wno-psabi -Wno-error=old-style-cast -faligned-new -Wno-maybe-uninitialized -fno-math-errno -fno-trapping-math -Werror=format -Wno-dangling-reference -Wno-error=dangling-reference -Wno-stringop-overflow, LAPACK_INFO=mkl, PERF_WITH_AVX=1, PERF_WITH_AVX2=1, TORCH_VERSION=2.12.0, USE_CUDA=0, USE_CUDNN=OFF, USE_CUSPARSELT=OFF, USE_GFLAGS=OFF, USE_GLOG=OFF, USE_GLOO=ON, USE_MKL=ON, USE_MKLDNN=ON, USE_MPI=OFF, USE_NCCL=OFF, USE_NNPACK=ON, USE_OPENMP=ON, USE_ROCM=OFF, USE_ROCM_KERNEL_ASSERT=OFF, USE_XCCL=OFF, USE_XPU=OFF, \n",
  "snac_load_seconds": 3.7510019540004578,
  "server_command": [
    "/tmp/navi-orpheus-bench.yjftjx/bin/llama-server",
    "-m",
    "/tmp/navi-orpheus-bench.yjftjx/Orpheus-3b-FT-Q4_K_M.gguf",
    "--host",
    "127.0.0.1",
    "--port",
    "15010",
    "-c",
    "8192",
    "-np",
    "1",
    "-ngl",
    "99",
    "-t",
    "1",
    "--rope-scaling",
    "linear",
    "--jinja",
    "-fa",
    "off",
    "--poll",
    "0",
    "--cache-ram",
    "0"
  ],
  "server_ready_seconds": 1.2608056370008853
}
