{
  "schemaVersion": 1,
  "startedAt": "2026-09-19T14:41:14.111Z",
  "pack": "librispeech-six-clip-starter-v1",
  "sampleManifestSha256": "65ca61b68897992a721a62cbc268a38e02b77224864e67375e8ee3407a94ca4d",
  "hardware": {
    "cpu": "13th Gen Intel(R) Core(TM) i9-13980HX",
    "logicalProcessors": 32,
    "ramBytes": 68319281152,
    "os": "Windows_NT 10.0.26200 x64",
    "gpuAtStart": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5407 MiB, 1 %"
  },
  "measurement": {
    "repeats": 3,
    "threads": 8,
    "timeoutMs": 90000,
    "timing": "Process wall time, including process start, model load, decoding and shutdown. No explicit cache flush or warm-up. New process per attempt, normal desktop background load.",
    "decoding": "English, beam size 5, best-of 5, no timestamps. CPU forced with -ng; GPU use must be confirmed from engine logs.",
    "ordering": "Clip, then model, then repeat; backend order reversed on even repeats to reduce fixed ordering bias.",
    "scoring": "Word edit distance on first attempt per clip, NFKC lowercase, punctuation removed except apostrophes. WER can exceed 100%. Failed attempts reported separately.",
    "limits": "Six English audiobook clips, one computer. Not app end-to-end latency, microphone reliability, multilingual accuracy, a hardware recommendation or a competitor comparison."
  },
  "binary": {
    "name": "whisper-cli.exe",
    "sha256": "a4b3d3af180c4ac63d13dd58637fc12410fc828ad9b82ed230813eaf0d9c3dbc",
    "components": [
      {
        "name": "concrt140.dll",
        "sha256": "8be3396e1811a4a021e600a9ec0a2f22cf44ed3ca237e0225ad97935db00dbaa"
      },
      {
        "name": "cublas64_11.dll",
        "sha256": "fb8eac115a5a9f82527baf7ccc1d81d56ea65c7661eb1359009da30551bb4b61"
      },
      {
        "name": "cublas64_12.dll",
        "sha256": "1e10d47d4bf4eb3695a9724a949b57b3737df95b58c21b7830fa826a65310a5a"
      },
      {
        "name": "cublasLt64_11.dll",
        "sha256": "e082070a79be1f041aefd5cf3d09053c612861eac6468dd021a6e7ea019d05f1"
      },
      {
        "name": "cublasLt64_12.dll",
        "sha256": "0cfc68a00c56aa54b62d76492b75b114be713672cfbc15fa6cd6047945b5a37b"
      },
      {
        "name": "cudart64_110.dll",
        "sha256": "dde19a1ad600689df4b7c21f6e379020f049100ca2910cfb3c2ea0e85b2168ad"
      },
      {
        "name": "cudart64_12.dll",
        "sha256": "3a1f9b1f8b119d66801cd248241efe93cd7cacfd548ea83134514b7b862642c9"
      },
      {
        "name": "ggml-base.dll",
        "sha256": "103741f0623f3fb6d27f0b890bb2a1a58590298febcad9a94d831ffdd2b87750"
      },
      {
        "name": "ggml-cpu.dll",
        "sha256": "017fb3ed2ef0097218b492b18305628d34190b61fe4b97dc5edea2952148734b"
      },
      {
        "name": "ggml-cuda.dll",
        "sha256": "8ba16d8754bad4d5def3d744ccbb297104314e80cc5c2c6b1e43fca74de622da"
      },
      {
        "name": "ggml.dll",
        "sha256": "efd350c07a0dac57a1c0a4289c5e6a7d566f78313a8d01538f29c9997511c224"
      },
      {
        "name": "msvcp140.dll",
        "sha256": "0f885b509a685d2bbfa652fed26b5fb31d88fbdab0a978c641d1c7b8aa460aa9"
      },
      {
        "name": "msvcp140_1.dll",
        "sha256": "ad99d6781e3686512cf3d52c7dadc9b3f799dda7c522488e0f81e8272a492ed3"
      },
      {
        "name": "msvcp140_2.dll",
        "sha256": "04aaf7d461a49b4c2f02f6eb4289cadb926abb56f35ade07a0f9eacb5988944c"
      },
      {
        "name": "msvcp140_atomic_wait.dll",
        "sha256": "625a7a73a0ad8ccd90f84294315785c4f46e3378c98ed724e863072882930f8b"
      },
      {
        "name": "msvcp140_codecvt_ids.dll",
        "sha256": "d3017430a5ae6ec04a3f20995f177f9b0df062ceac8fbb495a8327b3e86be584"
      },
      {
        "name": "SDL2.dll",
        "sha256": "bf47e6ab76294518d64936127b530c238dd08277e03b76e0c20231340bc57285"
      },
      {
        "name": "vccorlib140.dll",
        "sha256": "e6b2140536b335624d68babe5b5ae913b45e73c54a564371234660d1901b6b52"
      },
      {
        "name": "vcomp140.dll",
        "sha256": "651c76dcd8a4b3ddbe63eed846e5e3688755b70445f0e9c56200fafa80161bd8"
      },
      {
        "name": "vcruntime140.dll",
        "sha256": "d5e4d9a3e835fa679450145d6a7d94e36573a509317111904d9b3712c30d9066"
      },
      {
        "name": "vcruntime140_1.dll",
        "sha256": "1f2d41c4aa5db0bc33ebf7b66d72943a817d7ce6cbe880502a9403823633093f"
      },
      {
        "name": "whisper-cli.exe",
        "sha256": "a4b3d3af180c4ac63d13dd58637fc12410fc828ad9b82ed230813eaf0d9c3dbc"
      },
      {
        "name": "whisper.dll",
        "sha256": "1b194a902b42e41f52bc9c28a230d7450a6b78a68fe82edb67cccacb6576e2c6"
      }
    ]
  },
  "models": [
    {
      "name": "base",
      "sha256": "60ed5bc3dd14eea856493d334349b405782ddcaf0028d4b5df4088345fba2efe"
    },
    {
      "name": "small",
      "sha256": "1be3a9b2063867b937e64e2ec7483364a79917e157fa98c5d94b5c1fffea987b"
    }
  ],
  "runs": [
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "base",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 3.505,
      "wallMs": 1532,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5407 MiB, 1 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Concorde returned to its place amidst the tents.",
      "wer": {
        "errors": 1,
        "referenceWords": 8,
        "rate": 0.125
      },
      "stdout": "\r\n Concorde returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   150.76 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.65 ms\r\nwhisper_print_timings:   sample time =    67.09 ms /    60 runs (     1.12 ms per run)\r\nwhisper_print_timings:   encode time =   712.04 ms /     1 runs (   712.04 ms per run)\r\nwhisper_print_timings:   decode time =     3.96 ms /     1 runs (     3.96 ms per run)\r\nwhisper_print_timings:   batchd time =    95.86 ms /    58 runs (     1.65 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1059.25 ms\r\n"
    },
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "base",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 3.505,
      "wallMs": 536,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5402 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Concorde returned to its place amidst the tents.",
      "wer": {
        "errors": 1,
        "referenceWords": 8,
        "rate": 0.125
      },
      "stdout": "\r\n Concorde returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   209.63 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     3.17 ms\r\nwhisper_print_timings:   sample time =    23.01 ms /    60 runs (     0.38 ms per run)\r\nwhisper_print_timings:   encode time =    75.43 ms /     1 runs (    75.43 ms per run)\r\nwhisper_print_timings:   decode time =     6.31 ms /     1 runs (     6.31 ms per run)\r\nwhisper_print_timings:   batchd time =    41.80 ms /    58 runs (     0.72 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   369.37 ms\r\n"
    },
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "base",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 3.505,
      "wallMs": 471,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5405 MiB, 7 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Concorde returned to its place amidst the tents.",
      "wer": {
        "errors": 1,
        "referenceWords": 8,
        "rate": 0.125
      },
      "stdout": "\r\n Concorde returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   184.26 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     2.33 ms\r\nwhisper_print_timings:   sample time =    18.46 ms /    60 runs (     0.31 ms per run)\r\nwhisper_print_timings:   encode time =    74.49 ms /     1 runs (    74.49 ms per run)\r\nwhisper_print_timings:   decode time =     5.52 ms /     1 runs (     5.52 ms per run)\r\nwhisper_print_timings:   batchd time =    43.06 ms /    58 runs (     0.74 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   337.59 ms\r\n"
    },
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "base",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 3.505,
      "wallMs": 1041,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5266 MiB, 10 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Concorde returned to its place amidst the tents.",
      "wer": {
        "errors": 1,
        "referenceWords": 8,
        "rate": 0.125
      },
      "stdout": "\r\n Concorde returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   110.11 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     2.53 ms\r\nwhisper_print_timings:   sample time =    65.80 ms /    60 runs (     1.10 ms per run)\r\nwhisper_print_timings:   encode time =   629.33 ms /     1 runs (   629.33 ms per run)\r\nwhisper_print_timings:   decode time =     4.85 ms /     1 runs (     4.85 ms per run)\r\nwhisper_print_timings:   batchd time =    80.34 ms /    58 runs (     1.39 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   914.24 ms\r\n"
    },
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "base",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 3.505,
      "wallMs": 1043,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5267 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Concorde returned to its place amidst the tents.",
      "wer": {
        "errors": 1,
        "referenceWords": 8,
        "rate": 0.125
      },
      "stdout": "\r\n Concorde returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   111.64 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     2.49 ms\r\nwhisper_print_timings:   sample time =    61.94 ms /    60 runs (     1.03 ms per run)\r\nwhisper_print_timings:   encode time =   637.61 ms /     1 runs (   637.61 ms per run)\r\nwhisper_print_timings:   decode time =     3.92 ms /     1 runs (     3.92 ms per run)\r\nwhisper_print_timings:   batchd time =    81.79 ms /    58 runs (     1.41 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   922.13 ms\r\n"
    },
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "base",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 3.505,
      "wallMs": 569,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5266 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Concorde returned to its place amidst the tents.",
      "wer": {
        "errors": 1,
        "referenceWords": 8,
        "rate": 0.125
      },
      "stdout": "\r\n Concorde returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   177.21 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     2.56 ms\r\nwhisper_print_timings:   sample time =    20.76 ms /    60 runs (     0.35 ms per run)\r\nwhisper_print_timings:   encode time =    65.38 ms /     1 runs (    65.38 ms per run)\r\nwhisper_print_timings:   decode time =     5.09 ms /     1 runs (     5.09 ms per run)\r\nwhisper_print_timings:   batchd time =    38.34 ms /    58 runs (     0.66 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   319.75 ms\r\n"
    },
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "small",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 3.505,
      "wallMs": 5080,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5167 MiB, 26 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Concord returned to its place amidst the tents.",
      "wer": {
        "errors": 0,
        "referenceWords": 8,
        "rate": 0
      },
      "stdout": "\r\n Concord returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   328.74 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     2.34 ms\r\nwhisper_print_timings:   sample time =   150.83 ms /    60 runs (     2.51 ms per run)\r\nwhisper_print_timings:   encode time =  2951.74 ms /     1 runs (  2951.74 ms per run)\r\nwhisper_print_timings:   decode time =    27.79 ms /     1 runs (    27.79 ms per run)\r\nwhisper_print_timings:   batchd time =  1128.04 ms /    58 runs (    19.45 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  4699.83 ms\r\n"
    },
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "small",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 3.505,
      "wallMs": 866,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5140 MiB, 15 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Concord returned to its place amidst the tents.",
      "wer": {
        "errors": 0,
        "referenceWords": 8,
        "rate": 0
      },
      "stdout": "\r\n Concord returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   406.09 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     2.42 ms\r\nwhisper_print_timings:   sample time =    18.17 ms /    60 runs (     0.30 ms per run)\r\nwhisper_print_timings:   encode time =    98.05 ms /     1 runs (    98.05 ms per run)\r\nwhisper_print_timings:   decode time =    11.70 ms /     1 runs (    11.70 ms per run)\r\nwhisper_print_timings:   batchd time =    77.33 ms /    58 runs (     1.33 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   627.75 ms\r\n"
    },
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "small",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 3.505,
      "wallMs": 765,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5164 MiB, 13 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Concord returned to its place amidst the tents.",
      "wer": {
        "errors": 0,
        "referenceWords": 8,
        "rate": 0
      },
      "stdout": "\r\n Concord returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   418.88 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     2.92 ms\r\nwhisper_print_timings:   sample time =    18.85 ms /    60 runs (     0.31 ms per run)\r\nwhisper_print_timings:   encode time =    91.44 ms /     1 runs (    91.44 ms per run)\r\nwhisper_print_timings:   decode time =     9.30 ms /     1 runs (     9.30 ms per run)\r\nwhisper_print_timings:   batchd time =    64.38 ms /    58 runs (     1.11 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   618.90 ms\r\n"
    },
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "small",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 3.505,
      "wallMs": 3742,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5165 MiB, 46 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Concord returned to its place amidst the tents.",
      "wer": {
        "errors": 0,
        "referenceWords": 8,
        "rate": 0
      },
      "stdout": "\r\n Concord returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   331.39 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     2.59 ms\r\nwhisper_print_timings:   sample time =    99.46 ms /    60 runs (     1.66 ms per run)\r\nwhisper_print_timings:   encode time =  2351.86 ms /     1 runs (  2351.86 ms per run)\r\nwhisper_print_timings:   decode time =    30.00 ms /     1 runs (    30.00 ms per run)\r\nwhisper_print_timings:   batchd time =   596.14 ms /    58 runs (    10.28 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  3464.77 ms\r\n"
    },
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "small",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 3.505,
      "wallMs": 4414,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5163 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Concord returned to its place amidst the tents.",
      "wer": {
        "errors": 0,
        "referenceWords": 8,
        "rate": 0
      },
      "stdout": "\r\n Concord returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   674.24 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.48 ms\r\nwhisper_print_timings:   sample time =    82.28 ms /    60 runs (     1.37 ms per run)\r\nwhisper_print_timings:   encode time =  3050.37 ms /     1 runs (  3050.37 ms per run)\r\nwhisper_print_timings:   decode time =    12.85 ms /     1 runs (    12.85 ms per run)\r\nwhisper_print_timings:   batchd time =   270.03 ms /    58 runs (     4.66 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  4168.47 ms\r\n"
    },
    {
      "clip": "6930-75918-0000",
      "split": "test-clean",
      "model": "small",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 3.505,
      "wallMs": 1169,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5237 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Concord returned to its place amidst the tents.",
      "wer": {
        "errors": 0,
        "referenceWords": 8,
        "rate": 0
      },
      "stdout": "\r\n Concord returned to its place amidst the tents.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-6930-75918-0000.wav' (56080 samples, 3.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   687.11 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     5.95 ms\r\nwhisper_print_timings:   sample time =    53.84 ms /    60 runs (     0.90 ms per run)\r\nwhisper_print_timings:   encode time =   108.71 ms /     1 runs (   108.71 ms per run)\r\nwhisper_print_timings:   decode time =     9.42 ms /     1 runs (     9.42 ms per run)\r\nwhisper_print_timings:   batchd time =   103.53 ms /    58 runs (     1.78 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   987.30 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "base",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 6.285,
      "wallMs": 1508,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5141 MiB, 44 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   173.57 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.90 ms\r\nwhisper_print_timings:   sample time =   146.93 ms /   124 runs (     1.18 ms per run)\r\nwhisper_print_timings:   encode time =   808.62 ms /     1 runs (   808.62 ms per run)\r\nwhisper_print_timings:   decode time =     4.49 ms /     1 runs (     4.49 ms per run)\r\nwhisper_print_timings:   batchd time =   191.56 ms /   122 runs (     1.57 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1360.08 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "base",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 6.285,
      "wallMs": 629,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5126 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   194.94 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.14 ms\r\nwhisper_print_timings:   sample time =    88.09 ms /   124 runs (     0.71 ms per run)\r\nwhisper_print_timings:   encode time =    70.76 ms /     1 runs (    70.76 ms per run)\r\nwhisper_print_timings:   decode time =     8.54 ms /     1 runs (     8.54 ms per run)\r\nwhisper_print_timings:   batchd time =    93.98 ms /   122 runs (     0.77 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   473.85 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "base",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 6.285,
      "wallMs": 674,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4999 MiB, 20 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   235.85 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.13 ms\r\nwhisper_print_timings:   sample time =    72.42 ms /   124 runs (     0.58 ms per run)\r\nwhisper_print_timings:   encode time =    80.42 ms /     1 runs (    80.42 ms per run)\r\nwhisper_print_timings:   decode time =     6.35 ms /     1 runs (     6.35 ms per run)\r\nwhisper_print_timings:   batchd time =    89.85 ms /   122 runs (     0.74 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   502.62 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "base",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 6.285,
      "wallMs": 1482,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5005 MiB, 35 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   154.50 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     5.16 ms\r\nwhisper_print_timings:   sample time =   152.81 ms /   124 runs (     1.23 ms per run)\r\nwhisper_print_timings:   encode time =   783.48 ms /     1 runs (   783.48 ms per run)\r\nwhisper_print_timings:   decode time =     4.32 ms /     1 runs (     4.32 ms per run)\r\nwhisper_print_timings:   batchd time =   199.78 ms /   122 runs (     1.64 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1325.31 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "base",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 6.285,
      "wallMs": 1459,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5002 MiB, 9 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   116.79 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     3.50 ms\r\nwhisper_print_timings:   sample time =   146.37 ms /   124 runs (     1.18 ms per run)\r\nwhisper_print_timings:   encode time =   831.29 ms /     1 runs (   831.29 ms per run)\r\nwhisper_print_timings:   decode time =     5.71 ms /     1 runs (     5.71 ms per run)\r\nwhisper_print_timings:   batchd time =   184.09 ms /   122 runs (     1.51 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1311.48 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "base",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 6.285,
      "wallMs": 592,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4987 MiB, 9 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout, the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   196.51 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     3.76 ms\r\nwhisper_print_timings:   sample time =    56.76 ms /   124 runs (     0.46 ms per run)\r\nwhisper_print_timings:   encode time =    70.11 ms /     1 runs (    70.11 ms per run)\r\nwhisper_print_timings:   decode time =     6.60 ms /     1 runs (     6.60 ms per run)\r\nwhisper_print_timings:   batchd time =    92.00 ms /   122 runs (     0.75 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   448.14 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "small",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 6.285,
      "wallMs": 4048,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4986 MiB, 23 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   545.49 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.40 ms\r\nwhisper_print_timings:   sample time =   143.35 ms /   121 runs (     1.18 ms per run)\r\nwhisper_print_timings:   encode time =  2605.68 ms /     1 runs (  2605.68 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   487.29 ms /   120 runs (     4.06 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  3852.50 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "small",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 6.285,
      "wallMs": 936,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4991 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   484.00 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.67 ms\r\nwhisper_print_timings:   sample time =    43.23 ms /   121 runs (     0.36 ms per run)\r\nwhisper_print_timings:   encode time =    94.84 ms /     1 runs (    94.84 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   130.48 ms /   120 runs (     1.09 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   771.29 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "small",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 6.285,
      "wallMs": 854,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4995 MiB, 34 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   422.41 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.13 ms\r\nwhisper_print_timings:   sample time =    40.79 ms /   121 runs (     0.34 ms per run)\r\nwhisper_print_timings:   encode time =    90.91 ms /     1 runs (    90.91 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   132.04 ms /   120 runs (     1.10 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   717.16 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "small",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 6.285,
      "wallMs": 3519,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4992 MiB, 14 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   354.13 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     3.45 ms\r\nwhisper_print_timings:   sample time =   143.57 ms /   121 runs (     1.19 ms per run)\r\nwhisper_print_timings:   encode time =  2284.44 ms /     1 runs (  2284.44 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   527.71 ms /   120 runs (     4.40 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  3375.65 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "small",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 6.285,
      "wallMs": 12370,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4974 MiB, 2 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   368.31 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     5.15 ms\r\nwhisper_print_timings:   sample time =   285.14 ms /   121 runs (     2.36 ms per run)\r\nwhisper_print_timings:   encode time =  9017.30 ms /     1 runs (  9017.30 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =  2280.65 ms /   120 runs (    19.01 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time = 12083.02 ms\r\n"
    },
    {
      "clip": "1320-122617-0003",
      "split": "test-clean",
      "model": "small",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 6.285,
      "wallMs": 807,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4915 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n There was something in his air and manner that betrayed to the scout the utter confusion of the state of his mind.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-1320-122617-0003.wav' (100560 samples, 6.3 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   398.15 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     3.26 ms\r\nwhisper_print_timings:   sample time =    33.73 ms /   121 runs (     0.28 ms per run)\r\nwhisper_print_timings:   encode time =    80.69 ms /     1 runs (    80.69 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   116.53 ms /   120 runs (     0.97 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   645.78 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "base",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 17.43,
      "wallMs": 1450,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4885 MiB, 16 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   106.32 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     7.45 ms\r\nwhisper_print_timings:   sample time =   330.41 ms /   338 runs (     0.98 ms per run)\r\nwhisper_print_timings:   encode time =   532.83 ms /     1 runs (   532.83 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   318.74 ms /   337 runs (     0.95 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1318.90 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "base",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 17.43,
      "wallMs": 670,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4884 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   170.55 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.42 ms\r\nwhisper_print_timings:   sample time =    98.38 ms /   338 runs (     0.29 ms per run)\r\nwhisper_print_timings:   encode time =    61.76 ms /     1 runs (    61.76 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   181.05 ms /   337 runs (     0.54 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   527.94 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "base",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 17.43,
      "wallMs": 653,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4875 MiB, 31 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   169.26 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     7.32 ms\r\nwhisper_print_timings:   sample time =    97.03 ms /   338 runs (     0.29 ms per run)\r\nwhisper_print_timings:   encode time =    61.14 ms /     1 runs (    61.14 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   173.93 ms /   337 runs (     0.52 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   518.13 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "base",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 17.43,
      "wallMs": 1494,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4875 MiB, 22 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   111.27 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.92 ms\r\nwhisper_print_timings:   sample time =   335.73 ms /   338 runs (     0.99 ms per run)\r\nwhisper_print_timings:   encode time =   563.47 ms /     1 runs (   563.47 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   333.28 ms /   337 runs (     0.99 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1373.64 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "base",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 17.43,
      "wallMs": 1701,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4847 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   132.79 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     9.22 ms\r\nwhisper_print_timings:   sample time =   345.17 ms /   338 runs (     1.02 ms per run)\r\nwhisper_print_timings:   encode time =   613.89 ms /     1 runs (   613.89 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   448.99 ms /   337 runs (     1.33 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1577.02 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "base",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 17.43,
      "wallMs": 644,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4864 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill, the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   168.91 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     7.49 ms\r\nwhisper_print_timings:   sample time =    89.16 ms /   338 runs (     0.26 ms per run)\r\nwhisper_print_timings:   encode time =    63.73 ms /     1 runs (    63.73 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   175.02 ms /   337 runs (     0.52 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   513.62 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "small",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 17.43,
      "wallMs": 10346,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4859 MiB, 17 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   315.02 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     7.62 ms\r\nwhisper_print_timings:   sample time =   644.78 ms /   340 runs (     1.90 ms per run)\r\nwhisper_print_timings:   encode time =  1914.14 ms /     1 runs (  1914.14 ms per run)\r\nwhisper_print_timings:   decode time =    39.08 ms /     1 runs (    39.08 ms per run)\r\nwhisper_print_timings:   batchd time =  7104.83 ms /   338 runs (    21.02 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time = 10077.97 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "small",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 17.43,
      "wallMs": 1083,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4805 MiB, 4 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   397.39 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.42 ms\r\nwhisper_print_timings:   sample time =    97.82 ms /   340 runs (     0.29 ms per run)\r\nwhisper_print_timings:   encode time =    87.63 ms /     1 runs (    87.63 ms per run)\r\nwhisper_print_timings:   decode time =     7.26 ms /     1 runs (     7.26 ms per run)\r\nwhisper_print_timings:   batchd time =   336.90 ms /   338 runs (     1.00 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   948.13 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "small",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 17.43,
      "wallMs": 1076,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4798 MiB, 59 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   413.38 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.92 ms\r\nwhisper_print_timings:   sample time =    96.81 ms /   340 runs (     0.28 ms per run)\r\nwhisper_print_timings:   encode time =    86.78 ms /     1 runs (    86.78 ms per run)\r\nwhisper_print_timings:   decode time =     8.56 ms /     1 runs (     8.56 ms per run)\r\nwhisper_print_timings:   batchd time =   315.14 ms /   338 runs (     0.93 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   939.65 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "small",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 17.43,
      "wallMs": 9202,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4797 MiB, 48 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   346.35 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     7.74 ms\r\nwhisper_print_timings:   sample time =   666.87 ms /   340 runs (     1.96 ms per run)\r\nwhisper_print_timings:   encode time =  2037.94 ms /     1 runs (  2037.94 ms per run)\r\nwhisper_print_timings:   decode time =    32.43 ms /     1 runs (    32.43 ms per run)\r\nwhisper_print_timings:   batchd time =  5772.26 ms /   338 runs (    17.08 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  8923.79 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "small",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 17.43,
      "wallMs": 9773,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4797 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   322.20 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =    10.40 ms\r\nwhisper_print_timings:   sample time =   723.18 ms /   340 runs (     2.13 ms per run)\r\nwhisper_print_timings:   encode time =  2383.14 ms /     1 runs (  2383.14 ms per run)\r\nwhisper_print_timings:   decode time =    28.38 ms /     1 runs (    28.38 ms per run)\r\nwhisper_print_timings:   batchd time =  5989.41 ms /   338 runs (    17.72 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  9513.64 ms\r\n"
    },
    {
      "clip": "5639-40744-0032",
      "split": "test-clean",
      "model": "small",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 17.43,
      "wallMs": 1237,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4797 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "wer": {
        "errors": 2,
        "referenceWords": 55,
        "rate": 0.03636363636363636
      },
      "stdout": "\r\n For God's sake, my lady mother, give me a wife who would be an agreeable companion, not one who will disgust me so that we may both bear evenly and with mutual goodwill the yoke imposed on us by heaven, instead of pulling this way and that way and fretting each other to death.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-clean-5639-40744-0032.wav' (278880 samples, 17.4 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   444.90 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     8.83 ms\r\nwhisper_print_timings:   sample time =    97.71 ms /   340 runs (     0.29 ms per run)\r\nwhisper_print_timings:   encode time =    86.41 ms /     1 runs (    86.41 ms per run)\r\nwhisper_print_timings:   decode time =     7.15 ms /     1 runs (     7.15 ms per run)\r\nwhisper_print_timings:   batchd time =   340.56 ms /   338 runs (     1.01 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1000.76 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 14.73,
      "wallMs": 1664,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4797 MiB, 27 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "wer": {
        "errors": 9,
        "referenceWords": 50,
        "rate": 0.18
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   109.60 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.63 ms\r\nwhisper_print_timings:   sample time =   352.00 ms /   321 runs (     1.10 ms per run)\r\nwhisper_print_timings:   encode time =   609.56 ms /     1 runs (   609.56 ms per run)\r\nwhisper_print_timings:   decode time =     7.16 ms /     2 runs (     3.58 ms per run)\r\nwhisper_print_timings:   batchd time =   429.13 ms /   318 runs (     1.35 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1541.14 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 14.73,
      "wallMs": 687,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4797 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "wer": {
        "errors": 9,
        "referenceWords": 50,
        "rate": 0.18
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   173.43 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.83 ms\r\nwhisper_print_timings:   sample time =    98.46 ms /   321 runs (     0.31 ms per run)\r\nwhisper_print_timings:   encode time =    68.92 ms /     1 runs (    68.92 ms per run)\r\nwhisper_print_timings:   decode time =     6.95 ms /     2 runs (     3.47 ms per run)\r\nwhisper_print_timings:   batchd time =   178.23 ms /   318 runs (     0.56 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   542.38 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 14.73,
      "wallMs": 785,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4797 MiB, 12 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "wer": {
        "errors": 9,
        "referenceWords": 50,
        "rate": 0.18
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   182.84 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.30 ms\r\nwhisper_print_timings:   sample time =    94.32 ms /   321 runs (     0.29 ms per run)\r\nwhisper_print_timings:   encode time =    63.97 ms /     1 runs (    63.97 ms per run)\r\nwhisper_print_timings:   decode time =    12.45 ms /     2 runs (     6.23 ms per run)\r\nwhisper_print_timings:   batchd time =   180.96 ms /   318 runs (     0.57 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   550.61 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 14.73,
      "wallMs": 1668,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4797 MiB, 15 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "wer": {
        "errors": 9,
        "referenceWords": 50,
        "rate": 0.18
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   134.15 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     9.29 ms\r\nwhisper_print_timings:   sample time =   323.92 ms /   321 runs (     1.01 ms per run)\r\nwhisper_print_timings:   encode time =   609.74 ms /     1 runs (   609.74 ms per run)\r\nwhisper_print_timings:   decode time =     8.08 ms /     2 runs (     4.04 ms per run)\r\nwhisper_print_timings:   batchd time =   431.60 ms /   318 runs (     1.36 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1542.65 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 14.73,
      "wallMs": 1425,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4797 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "wer": {
        "errors": 9,
        "referenceWords": 50,
        "rate": 0.18
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   109.11 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.61 ms\r\nwhisper_print_timings:   sample time =   311.23 ms /   321 runs (     0.97 ms per run)\r\nwhisper_print_timings:   encode time =   512.98 ms /     1 runs (   512.98 ms per run)\r\nwhisper_print_timings:   decode time =     5.86 ms /     2 runs (     2.93 ms per run)\r\nwhisper_print_timings:   batchd time =   321.29 ms /   318 runs (     1.01 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1289.71 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 14.73,
      "wallMs": 657,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4797 MiB, 3 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "wer": {
        "errors": 9,
        "referenceWords": 50,
        "rate": 0.18
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of excitation, injured and more proper, as the Prince called our love of our own dignity, of which Archibald Ray Stroke and the full flush of his young belief in his importance as a British officer had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   171.76 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.86 ms\r\nwhisper_print_timings:   sample time =    93.90 ms /   321 runs (     0.29 ms per run)\r\nwhisper_print_timings:   encode time =    63.37 ms /     1 runs (    63.37 ms per run)\r\nwhisper_print_timings:   decode time =     8.79 ms /     2 runs (     4.40 ms per run)\r\nwhisper_print_timings:   batchd time =   172.71 ms /   318 runs (     0.54 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   526.95 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 14.73,
      "wallMs": 7824,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4797 MiB, 25 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "wer": {
        "errors": 5,
        "referenceWords": 50,
        "rate": 0.1
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   336.28 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.44 ms\r\nwhisper_print_timings:   sample time =   599.11 ms /   339 runs (     1.77 ms per run)\r\nwhisper_print_timings:   encode time =  2015.74 ms /     1 runs (  2015.74 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =  4543.96 ms /   338 runs (    13.44 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  7557.50 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 14.73,
      "wallMs": 1316,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4792 MiB, 8 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "wer": {
        "errors": 5,
        "referenceWords": 50,
        "rate": 0.1
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   437.80 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.87 ms\r\nwhisper_print_timings:   sample time =   112.20 ms /   339 runs (     0.33 ms per run)\r\nwhisper_print_timings:   encode time =    98.20 ms /     1 runs (    98.20 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   456.02 ms /   338 runs (     1.35 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1125.84 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 14.73,
      "wallMs": 1195,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4792 MiB, 42 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "wer": {
        "errors": 5,
        "referenceWords": 50,
        "rate": 0.1
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   422.05 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     7.33 ms\r\nwhisper_print_timings:   sample time =   119.18 ms /   339 runs (     0.35 ms per run)\r\nwhisper_print_timings:   encode time =    92.08 ms /     1 runs (    92.08 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   400.42 ms /   338 runs (     1.18 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1056.34 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 14.73,
      "wallMs": 9871,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4797 MiB, 48 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "wer": {
        "errors": 5,
        "referenceWords": 50,
        "rate": 0.1
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   344.25 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.74 ms\r\nwhisper_print_timings:   sample time =   761.19 ms /   339 runs (     2.25 ms per run)\r\nwhisper_print_timings:   encode time =  2432.03 ms /     1 runs (  2432.03 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =  5983.12 ms /   338 runs (    17.70 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  9587.20 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 14.73,
      "wallMs": 7131,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4815 MiB, 7 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "wer": {
        "errors": 5,
        "referenceWords": 50,
        "rate": 0.1
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   341.49 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.77 ms\r\nwhisper_print_timings:   sample time =   366.83 ms /   339 runs (     1.08 ms per run)\r\nwhisper_print_timings:   encode time =  4779.15 ms /     1 runs (  4779.15 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =  1433.83 ms /   338 runs (     4.24 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  6987.47 ms\r\n"
    },
    {
      "clip": "7902-96591-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 14.73,
      "wallMs": 1170,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5259 MiB, 1 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7902-96591-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "wer": {
        "errors": 5,
        "referenceWords": 50,
        "rate": 0.1
      },
      "stdout": "\r\n He laughed, but it was a curious kind of laugh, full of vexation. \"Injured am I proper,\" as the French call our love of our own dignity, of which Archibald Ray stroke, in the full flush of his young belief in his importance as a British officer, had a pretty good stock.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7902-96591-0008.wav' (235680 samples, 14.7 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   419.64 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.59 ms\r\nwhisper_print_timings:   sample time =   112.06 ms /   339 runs (     0.33 ms per run)\r\nwhisper_print_timings:   encode time =    90.28 ms /     1 runs (    90.28 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   375.06 ms /   338 runs (     1.11 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1018.74 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "base",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 7.54,
      "wallMs": 1218,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5075 MiB, 42 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 1,
        "referenceWords": 22,
        "rate": 0.045454545454545456
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   130.90 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.56 ms\r\nwhisper_print_timings:   sample time =   127.62 ms /   136 runs (     0.94 ms per run)\r\nwhisper_print_timings:   encode time =   647.70 ms /     1 runs (   647.70 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   157.12 ms /   135 runs (     1.16 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1092.94 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "base",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 7.54,
      "wallMs": 639,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5077 MiB, 10 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 1,
        "referenceWords": 22,
        "rate": 0.045454545454545456
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   297.44 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     3.74 ms\r\nwhisper_print_timings:   sample time =    35.28 ms /   136 runs (     0.26 ms per run)\r\nwhisper_print_timings:   encode time =    64.43 ms /     1 runs (    64.43 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =    85.98 ms /   135 runs (     0.64 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   495.72 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "base",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 7.54,
      "wallMs": 663,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5077 MiB, 42 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 1,
        "referenceWords": 22,
        "rate": 0.045454545454545456
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   176.90 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     3.54 ms\r\nwhisper_print_timings:   sample time =    44.63 ms /   136 runs (     0.33 ms per run)\r\nwhisper_print_timings:   encode time =    64.81 ms /     1 runs (    64.81 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   200.28 ms /   135 runs (     1.48 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   501.11 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "base",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 7.54,
      "wallMs": 1189,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5099 MiB, 26 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 1,
        "referenceWords": 22,
        "rate": 0.045454545454545456
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   120.80 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.22 ms\r\nwhisper_print_timings:   sample time =   137.71 ms /   136 runs (     1.01 ms per run)\r\nwhisper_print_timings:   encode time =   576.42 ms /     1 runs (   576.42 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   183.86 ms /   135 runs (     1.36 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1048.17 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "base",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 7.54,
      "wallMs": 1357,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5100 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 1,
        "referenceWords": 22,
        "rate": 0.045454545454545456
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   141.59 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.76 ms\r\nwhisper_print_timings:   sample time =   143.05 ms /   136 runs (     1.05 ms per run)\r\nwhisper_print_timings:   encode time =   716.06 ms /     1 runs (   716.06 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   191.17 ms /   135 runs (     1.42 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1220.77 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "base",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 7.54,
      "wallMs": 655,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4950 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 1,
        "referenceWords": 22,
        "rate": 0.045454545454545456
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all in everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   189.83 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.39 ms\r\nwhisper_print_timings:   sample time =   107.47 ms /   136 runs (     0.79 ms per run)\r\nwhisper_print_timings:   encode time =    80.15 ms /     1 runs (    80.15 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   100.86 ms /   135 runs (     0.75 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   494.34 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "small",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 7.54,
      "wallMs": 9184,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4902 MiB, 11 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   431.89 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     5.73 ms\r\nwhisper_print_timings:   sample time =   300.23 ms /   134 runs (     2.24 ms per run)\r\nwhisper_print_timings:   encode time =  5597.60 ms /     1 runs (  5597.60 ms per run)\r\nwhisper_print_timings:   decode time =    29.54 ms /     1 runs (    29.54 ms per run)\r\nwhisper_print_timings:   batchd time =  2381.65 ms /   132 runs (    18.04 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  8895.76 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "small",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 7.54,
      "wallMs": 877,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5170 MiB, 6 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   431.83 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.20 ms\r\nwhisper_print_timings:   sample time =    39.39 ms /   136 runs (     0.29 ms per run)\r\nwhisper_print_timings:   encode time =    90.88 ms /     1 runs (    90.88 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   151.19 ms /   135 runs (     1.12 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   730.52 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "small",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 7.54,
      "wallMs": 844,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5441 MiB, 41 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   415.21 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     3.06 ms\r\nwhisper_print_timings:   sample time =    39.96 ms /   136 runs (     0.29 ms per run)\r\nwhisper_print_timings:   encode time =    87.90 ms /     1 runs (    87.90 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   151.90 ms /   135 runs (     1.13 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   711.76 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "small",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 7.54,
      "wallMs": 5452,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5441 MiB, 13 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   439.31 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.46 ms\r\nwhisper_print_timings:   sample time =   275.50 ms /   134 runs (     2.06 ms per run)\r\nwhisper_print_timings:   encode time =  2349.51 ms /     1 runs (  2349.51 ms per run)\r\nwhisper_print_timings:   decode time =    11.51 ms /     1 runs (    11.51 ms per run)\r\nwhisper_print_timings:   batchd time =  2127.24 ms /   132 runs (    16.12 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  5302.42 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "small",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 7.54,
      "wallMs": 3609,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5168 MiB, 50 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   368.39 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     3.94 ms\r\nwhisper_print_timings:   sample time =   145.26 ms /   134 runs (     1.08 ms per run)\r\nwhisper_print_timings:   encode time =  2313.66 ms /     1 runs (  2313.66 ms per run)\r\nwhisper_print_timings:   decode time =    11.85 ms /     1 runs (    11.85 ms per run)\r\nwhisper_print_timings:   batchd time =   535.00 ms /   132 runs (     4.05 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  3451.69 ms\r\n"
    },
    {
      "clip": "7018-75788-0016",
      "split": "test-other",
      "model": "small",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 7.54,
      "wallMs": 882,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5106 MiB, 3 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-7018-75788-0016.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "wer": {
        "errors": 0,
        "referenceWords": 22,
        "rate": 0
      },
      "stdout": "\r\n Presently the ship struck the mountain and broke up, and all and everything on board of her were plunged into the sea.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-7018-75788-0016.wav' (120640 samples, 7.5 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   440.91 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     4.23 ms\r\nwhisper_print_timings:   sample time =    38.48 ms /   136 runs (     0.28 ms per run)\r\nwhisper_print_timings:   encode time =    85.54 ms /     1 runs (    85.54 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   145.68 ms /   135 runs (     1.08 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   728.44 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 18.81,
      "wallMs": 1569,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5130 MiB, 45 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "\"Arc you, my masters, you that love the wine.\" \"Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "wer": {
        "errors": 5,
        "referenceWords": 46,
        "rate": 0.10869565217391304
      },
      "stdout": "\r\n \"Arc you, my masters, you that love the wine.\" \"Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   114.87 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.20 ms\r\nwhisper_print_timings:   sample time =   304.15 ms /   295 runs (     1.03 ms per run)\r\nwhisper_print_timings:   encode time =   582.82 ms /     1 runs (   582.82 ms per run)\r\nwhisper_print_timings:   decode time =     3.87 ms /     1 runs (     3.87 ms per run)\r\nwhisper_print_timings:   batchd time =   418.20 ms /   293 runs (     1.43 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1453.02 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 18.81,
      "wallMs": 666,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5097 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "\"Arc you, my masters, you that love the wine.\" Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leaves to taste one drop of the liquor that will not now come and fight for relief of the vine.",
      "wer": {
        "errors": 6,
        "referenceWords": 46,
        "rate": 0.13043478260869565
      },
      "stdout": "\r\n \"Arc you, my masters, you that love the wine.\" Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leaves to taste one drop of the liquor that will not now come and fight for relief of the vine.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   179.77 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     7.80 ms\r\nwhisper_print_timings:   sample time =    89.18 ms /   295 runs (     0.30 ms per run)\r\nwhisper_print_timings:   encode time =    66.18 ms /     1 runs (    66.18 ms per run)\r\nwhisper_print_timings:   decode time =     5.39 ms /     1 runs (     5.39 ms per run)\r\nwhisper_print_timings:   batchd time =   172.06 ms /   293 runs (     0.59 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   530.40 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 18.81,
      "wallMs": 699,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4964 MiB, 14 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "\"Arc you, my masters, you that love the wine.\" Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leaves to taste one drop of the liquor that will not now come and fight for relief of the vine.",
      "wer": {
        "errors": 6,
        "referenceWords": 46,
        "rate": 0.13043478260869565
      },
      "stdout": "\r\n \"Arc you, my masters, you that love the wine.\" Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leaves to taste one drop of the liquor that will not now come and fight for relief of the vine.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   200.05 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     7.93 ms\r\nwhisper_print_timings:   sample time =    88.77 ms /   295 runs (     0.30 ms per run)\r\nwhisper_print_timings:   encode time =    75.07 ms /     1 runs (    75.07 ms per run)\r\nwhisper_print_timings:   decode time =     5.64 ms /     1 runs (     5.64 ms per run)\r\nwhisper_print_timings:   batchd time =   176.80 ms /   293 runs (     0.60 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   564.67 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 18.81,
      "wallMs": 1591,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4938 MiB, 32 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "\"Arc you, my masters, you that love the wine.\" \"Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "wer": {
        "errors": 5,
        "referenceWords": 46,
        "rate": 0.10869565217391304
      },
      "stdout": "\r\n \"Arc you, my masters, you that love the wine.\" \"Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   126.98 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     5.59 ms\r\nwhisper_print_timings:   sample time =   291.88 ms /   295 runs (     0.99 ms per run)\r\nwhisper_print_timings:   encode time =   605.92 ms /     1 runs (   605.92 ms per run)\r\nwhisper_print_timings:   decode time =     4.93 ms /     1 runs (     4.93 ms per run)\r\nwhisper_print_timings:   batchd time =   409.11 ms /   293 runs (     1.40 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1470.21 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 18.81,
      "wallMs": 3038,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4952 MiB, 0 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "\"Arc you, my masters, you that love the wine.\" \"Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "wer": {
        "errors": 5,
        "referenceWords": 46,
        "rate": 0.10869565217391304
      },
      "stdout": "\r\n \"Arc you, my masters, you that love the wine.\" \"Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   16.28 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   96.37 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   128.49 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.25 ms\r\nwhisper_print_timings:   sample time =   446.83 ms /   295 runs (     1.51 ms per run)\r\nwhisper_print_timings:   encode time =   583.54 ms /     1 runs (   583.54 ms per run)\r\nwhisper_print_timings:   decode time =     4.76 ms /     1 runs (     4.76 ms per run)\r\nwhisper_print_timings:   batchd time =  1679.13 ms /   293 runs (     5.73 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  2872.61 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "base",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 18.81,
      "wallMs": 1009,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5296 MiB, 1 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-base.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "\"Arc you, my masters, you that love the wine.\" Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leaves to taste one drop of the liquor that will not now come and fight for relief of the vine.",
      "wer": {
        "errors": 6,
        "referenceWords": 46,
        "rate": 0.13043478260869565
      },
      "stdout": "\r\n \"Arc you, my masters, you that love the wine.\" Cops body, follow me, for St. Anthony burned me as freely as a faggot. They get leaves to taste one drop of the liquor that will not now come and fight for relief of the vine.",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-base.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 512\r\nwhisper_model_load: n_audio_head  = 8\r\nwhisper_model_load: n_audio_layer = 6\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 512\r\nwhisper_model_load: n_text_head   = 8\r\nwhisper_model_load: n_text_layer  = 6\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 2 (base)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   147.37 MB\r\nwhisper_model_load: model size    =  147.37 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =    6.29 MB\r\nwhisper_init_state: kv cross size =   18.87 MB\r\nwhisper_init_state: kv pad  size  =    3.15 MB\r\nwhisper_init_state: compute buffer (conv)   =   17.24 MB\r\nwhisper_init_state: compute buffer (encode) =   85.88 MB\r\nwhisper_init_state: compute buffer (cross)  =    4.66 MB\r\nwhisper_init_state: compute buffer (decode) =   97.29 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   338.19 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =    15.71 ms\r\nwhisper_print_timings:   sample time =   127.46 ms /   295 runs (     0.43 ms per run)\r\nwhisper_print_timings:   encode time =   105.70 ms /     1 runs (   105.70 ms per run)\r\nwhisper_print_timings:   decode time =     5.75 ms /     1 runs (     5.75 ms per run)\r\nwhisper_print_timings:   batchd time =   185.18 ms /   293 runs (     0.63 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =   795.65 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cpu",
      "repeat": 1,
      "audioSeconds": 18.81,
      "wallMs": 10033,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5296 MiB, 17 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "\"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "wer": {
        "errors": 3,
        "referenceWords": 46,
        "rate": 0.06521739130434782
      },
      "stdout": "\r\n \"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   452.20 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     9.43 ms\r\nwhisper_print_timings:   sample time =   608.58 ms /   297 runs (     2.05 ms per run)\r\nwhisper_print_timings:   encode time =  3473.68 ms /     1 runs (  3473.68 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =  5117.32 ms /   296 runs (    17.29 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  9736.26 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cuda",
      "repeat": 1,
      "audioSeconds": 18.81,
      "wallMs": 1476,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 5074 MiB, 8 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "\"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "wer": {
        "errors": 3,
        "referenceWords": 46,
        "rate": 0.06521739130434782
      },
      "stdout": "\r\n \"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   472.23 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     8.29 ms\r\nwhisper_print_timings:   sample time =   190.47 ms /   297 runs (     0.64 ms per run)\r\nwhisper_print_timings:   encode time =   124.45 ms /     1 runs (   124.45 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   444.56 ms /   296 runs (     1.50 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1260.49 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cuda",
      "repeat": 2,
      "audioSeconds": 18.81,
      "wallMs": 1365,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4960 MiB, 46 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "\"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "wer": {
        "errors": 3,
        "referenceWords": 46,
        "rate": 0.06521739130434782
      },
      "stdout": "\r\n \"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   552.15 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =    12.06 ms\r\nwhisper_print_timings:   sample time =   121.78 ms /   297 runs (     0.41 ms per run)\r\nwhisper_print_timings:   encode time =   107.68 ms /     1 runs (   107.68 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   388.27 ms /   296 runs (     1.31 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1198.72 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cpu",
      "repeat": 2,
      "audioSeconds": 18.81,
      "wallMs": 15330,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4960 MiB, 81 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "\"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "wer": {
        "errors": 3,
        "referenceWords": 46,
        "rate": 0.06521739130434782
      },
      "stdout": "\r\n \"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   356.11 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     6.48 ms\r\nwhisper_print_timings:   sample time =   664.52 ms /   297 runs (     2.24 ms per run)\r\nwhisper_print_timings:   encode time =  8134.75 ms /     1 runs (  8134.75 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =  5756.88 ms /   296 runs (    19.45 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time = 15035.55 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cpu",
      "repeat": 3,
      "audioSeconds": 18.81,
      "wallMs": 12594,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4862 MiB, 10 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5",
        "-ng"
      ],
      "transcript": "\"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "wer": {
        "errors": 3,
        "referenceWords": 46,
        "rate": 0.06521739130434782
      },
      "stdout": "\r\n \"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 0\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:          CPU total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: no GPU found\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   22.42 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   97.28 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   351.16 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     8.68 ms\r\nwhisper_print_timings:   sample time =   672.04 ms /   297 runs (     2.26 ms per run)\r\nwhisper_print_timings:   encode time =  4649.57 ms /     1 runs (  4649.57 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =  6469.55 ms /   296 runs (    21.86 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time = 12278.28 ms\r\n"
    },
    {
      "clip": "4198-12281-0008",
      "split": "test-other",
      "model": "small",
      "backend": "cuda",
      "repeat": 3,
      "audioSeconds": 18.81,
      "wallMs": 1312,
      "gpuBefore": "NVIDIA GeForce RTX 4090 Laptop GPU, 610.88, 16376 MiB, 4839 MiB, 9 %",
      "backendVerified": true,
      "success": true,
      "exitCode": 0,
      "error": null,
      "command": [
        "-m",
        "<MODEL_DIR>\\ggml-small.bin",
        "-f",
        "<SAMPLE_DIR>\\test-other-4198-12281-0008.wav",
        "-l",
        "en",
        "-nt",
        "-t",
        "8",
        "-bs",
        "5",
        "-bo",
        "5"
      ],
      "transcript": "\"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "wer": {
        "errors": 3,
        "referenceWords": 46,
        "rate": 0.06521739130434782
      },
      "stdout": "\r\n \"Hark you, my masters, you that love the wine! Cops body, follow me, for St. Anthony burn me as freely as a faggot. They get leave to taste one drop of the liquor that will not now come and fight for relief of the vine.\"",
      "stderr": "ggml_cuda_init: GGML_CUDA_FORCE_MMQ:    no\r\nggml_cuda_init: GGML_CUDA_FORCE_CUBLAS: no\r\nggml_cuda_init: found 1 CUDA devices:\r\n  Device 0: NVIDIA GeForce RTX 4090 Laptop GPU, compute capability 8.9, VMM: yes\r\nwhisper_init_from_file_with_params_no_state: loading model from '<MODEL_DIR>\\ggml-small.bin'\r\nwhisper_init_with_params_no_state: use gpu    = 1\r\nwhisper_init_with_params_no_state: flash attn = 0\r\nwhisper_init_with_params_no_state: gpu_device = 0\r\nwhisper_init_with_params_no_state: dtw        = 0\r\nwhisper_init_with_params_no_state: devices    = 2\r\nwhisper_init_with_params_no_state: backends   = 2\r\nwhisper_model_load: loading model\r\nwhisper_model_load: n_vocab       = 51865\r\nwhisper_model_load: n_audio_ctx   = 1500\r\nwhisper_model_load: n_audio_state = 768\r\nwhisper_model_load: n_audio_head  = 12\r\nwhisper_model_load: n_audio_layer = 12\r\nwhisper_model_load: n_text_ctx    = 448\r\nwhisper_model_load: n_text_state  = 768\r\nwhisper_model_load: n_text_head   = 12\r\nwhisper_model_load: n_text_layer  = 12\r\nwhisper_model_load: n_mels        = 80\r\nwhisper_model_load: ftype         = 1\r\nwhisper_model_load: qntvr         = 0\r\nwhisper_model_load: type          = 3 (small)\r\nwhisper_model_load: adding 1608 extra tokens\r\nwhisper_model_load: n_langs       = 99\r\nwhisper_model_load:        CUDA0 total size =   487.01 MB\r\nwhisper_model_load: model size    =  487.01 MB\r\nwhisper_backend_init_gpu: using CUDA0 backend\r\nwhisper_init_state: kv self size  =   18.87 MB\r\nwhisper_init_state: kv cross size =   56.62 MB\r\nwhisper_init_state: kv pad  size  =    4.72 MB\r\nwhisper_init_state: compute buffer (conv)   =   23.38 MB\r\nwhisper_init_state: compute buffer (encode) =  128.02 MB\r\nwhisper_init_state: compute buffer (cross)  =    6.20 MB\r\nwhisper_init_state: compute buffer (decode) =   98.21 MB\r\n\r\nsystem_info: n_threads = 8 / 32 | WHISPER : COREML = 0 | OPENVINO = 0 | CUDA : ARCHS = 610,700,750,800,860,890 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | OPENMP = 1 | REPACK = 1 | \r\n\r\nmain: processing '<SAMPLE_DIR>\\test-other-4198-12281-0008.wav' (300960 samples, 18.8 sec), 8 threads, 1 processors, 5 beams + best of 5, lang = en, task = transcribe, timestamps = 0 ...\r\n\r\n\r\nwhisper_print_timings:     load time =   511.92 ms\r\nwhisper_print_timings:     fallbacks =   0 p /   0 h\r\nwhisper_print_timings:      mel time =     8.24 ms\r\nwhisper_print_timings:   sample time =   118.82 ms /   297 runs (     0.40 ms per run)\r\nwhisper_print_timings:   encode time =   105.89 ms /     1 runs (   105.89 ms per run)\r\nwhisper_print_timings:   decode time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:   batchd time =   353.51 ms /   296 runs (     1.19 ms per run)\r\nwhisper_print_timings:   prompt time =     0.00 ms /     1 runs (     0.00 ms per run)\r\nwhisper_print_timings:    total time =  1114.23 ms\r\n"
    }
  ],
  "summary": [
    {
      "configuration": "base/cpu",
      "attemptedRuns": 18,
      "successfulRuns": 18,
      "clips": [
        {
          "clip": "6930-75918-0000",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 1043,
          "minWallMs": 1041,
          "maxWallMs": 1532
        },
        {
          "clip": "1320-122617-0003",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 1482,
          "minWallMs": 1459,
          "maxWallMs": 1508
        },
        {
          "clip": "5639-40744-0032",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 1494,
          "minWallMs": 1450,
          "maxWallMs": 1701
        },
        {
          "clip": "7902-96591-0008",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 1664,
          "minWallMs": 1425,
          "maxWallMs": 1668
        },
        {
          "clip": "7018-75788-0016",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 1218,
          "minWallMs": 1189,
          "maxWallMs": 1357
        },
        {
          "clip": "4198-12281-0008",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 1591,
          "minWallMs": 1569,
          "maxWallMs": 3038
        }
      ],
      "medianWallMs": 1488,
      "firstAttemptClipsScored": 6,
      "firstAttemptFailures": 0,
      "firstAttemptWordErrors": 18,
      "firstAttemptReferenceWords": 203,
      "firstAttemptWer": 0.08866995073891626
    },
    {
      "configuration": "base/cuda",
      "attemptedRuns": 18,
      "successfulRuns": 18,
      "clips": [
        {
          "clip": "6930-75918-0000",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 536,
          "minWallMs": 471,
          "maxWallMs": 569
        },
        {
          "clip": "1320-122617-0003",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 629,
          "minWallMs": 592,
          "maxWallMs": 674
        },
        {
          "clip": "5639-40744-0032",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 653,
          "minWallMs": 644,
          "maxWallMs": 670
        },
        {
          "clip": "7902-96591-0008",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 687,
          "minWallMs": 657,
          "maxWallMs": 785
        },
        {
          "clip": "7018-75788-0016",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 655,
          "minWallMs": 639,
          "maxWallMs": 663
        },
        {
          "clip": "4198-12281-0008",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 699,
          "minWallMs": 666,
          "maxWallMs": 1009
        }
      ],
      "medianWallMs": 656,
      "firstAttemptClipsScored": 6,
      "firstAttemptFailures": 0,
      "firstAttemptWordErrors": 19,
      "firstAttemptReferenceWords": 203,
      "firstAttemptWer": 0.09359605911330049
    },
    {
      "configuration": "small/cpu",
      "attemptedRuns": 18,
      "successfulRuns": 18,
      "clips": [
        {
          "clip": "6930-75918-0000",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 4414,
          "minWallMs": 3742,
          "maxWallMs": 5080
        },
        {
          "clip": "1320-122617-0003",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 4048,
          "minWallMs": 3519,
          "maxWallMs": 12370
        },
        {
          "clip": "5639-40744-0032",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 9773,
          "minWallMs": 9202,
          "maxWallMs": 10346
        },
        {
          "clip": "7902-96591-0008",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 7824,
          "minWallMs": 7131,
          "maxWallMs": 9871
        },
        {
          "clip": "7018-75788-0016",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 5452,
          "minWallMs": 3609,
          "maxWallMs": 9184
        },
        {
          "clip": "4198-12281-0008",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 12594,
          "minWallMs": 10033,
          "maxWallMs": 15330
        }
      ],
      "medianWallMs": 8504,
      "firstAttemptClipsScored": 6,
      "firstAttemptFailures": 0,
      "firstAttemptWordErrors": 10,
      "firstAttemptReferenceWords": 203,
      "firstAttemptWer": 0.04926108374384237
    },
    {
      "configuration": "small/cuda",
      "attemptedRuns": 18,
      "successfulRuns": 18,
      "clips": [
        {
          "clip": "6930-75918-0000",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 866,
          "minWallMs": 765,
          "maxWallMs": 1169
        },
        {
          "clip": "1320-122617-0003",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 854,
          "minWallMs": 807,
          "maxWallMs": 936
        },
        {
          "clip": "5639-40744-0032",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 1083,
          "minWallMs": 1076,
          "maxWallMs": 1237
        },
        {
          "clip": "7902-96591-0008",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 1195,
          "minWallMs": 1170,
          "maxWallMs": 1316
        },
        {
          "clip": "7018-75788-0016",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 877,
          "minWallMs": 844,
          "maxWallMs": 882
        },
        {
          "clip": "4198-12281-0008",
          "successfulRuns": 3,
          "attemptedRuns": 3,
          "medianWallMs": 1365,
          "minWallMs": 1312,
          "maxWallMs": 1476
        }
      ],
      "medianWallMs": 1079.5,
      "firstAttemptClipsScored": 6,
      "firstAttemptFailures": 0,
      "firstAttemptWordErrors": 10,
      "firstAttemptReferenceWords": 203,
      "firstAttemptWer": 0.04926108374384237
    }
  ],
  "finishedAt": "2026-09-19T14:44:41.165Z"
}
