From 361cad93fb5d1f536f2a96c946a12408db4e2fa2 Mon Sep 17 00:00:00 2001 From: CJ Pais Date: Fri, 25 Sep 2026 11:54:08 +0800 Subject: [PATCH] better error code returns --- src/arch/canary/model.cpp | 53 +++++++----- src/arch/canary_qwen/model.cpp | 56 ++++++------ src/arch/cohere/model.cpp | 50 ++++++----- src/arch/funasr_nano/model.cpp | 50 ++++++----- src/arch/gigaam/model.cpp | 20 ++--- src/arch/granite/model.cpp | 33 ++++--- src/arch/granite5_ctc/model.cpp | 10 +-- src/arch/granite_nar/model.cpp | 10 +-- src/arch/medasr/model.cpp | 20 ++--- src/arch/moonshine/model.cpp | 28 +++--- src/arch/moonshine_streaming/model.cpp | 52 +++++------ src/arch/moss/model.cpp | 26 +++--- src/arch/parakeet/decoder.cpp | 46 +++++++--- src/arch/parakeet/model.cpp | 38 ++++----- src/arch/parakeet/multitalker.cpp | 2 +- src/arch/qwen3_asr/model.cpp | 34 ++++---- src/arch/sensevoice/model.cpp | 20 ++--- src/arch/sortformer/model.cpp | 24 +++--- src/arch/voxtral/model.cpp | 26 +++--- src/arch/voxtral_realtime/model.cpp | 114 ++++++++++++++----------- src/arch/whisper/bin_load.cpp | 4 +- src/arch/whisper/model.cpp | 96 ++++++++++----------- src/causal_lm/causal_lm.cpp | 30 +++---- src/causal_lm/causal_lm.h | 16 ++-- src/transcribe-batch-util.cpp | 48 ++++++----- src/transcribe-batch-util.h | 7 +- src/transcribe-load-common.cpp | 6 +- 27 files changed, 494 insertions(+), 425 deletions(-) diff --git a/src/arch/canary/model.cpp b/src/arch/canary/model.cpp index 775c1995..3581b7b4 100644 --- a/src/arch/canary/model.cpp +++ b/src/arch/canary/model.cpp @@ -289,7 +289,7 @@ transcribe_status fuse_batch_norm(CanaryModel & m) { ggml_init_params params = { ctx_size, nullptr, true }; m.bn_fused_ctx = ggml_init(params); if (m.bn_fused_ctx == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } for (size_t i = 0; i < n_blocks; ++i) { @@ -300,7 +300,7 @@ transcribe_status fuse_batch_norm(CanaryModel & m) { m.bn_fused_buffer = ggml_backend_alloc_ctx_tensors(m.bn_fused_ctx, m.plan.scheduler_list.back()); if (m.bn_fused_buffer == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } std::vector bn_w(d), bn_b(d), rm(d), rv(d); @@ -498,7 +498,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -779,7 +779,7 @@ transcribe_status run(transcribe_session * session, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary run: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -825,7 +825,7 @@ transcribe_status run(transcribe_session * session, const int64_t t_enc_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary run: encoder compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_encode_us = ggml_time_us() - t_enc_start; @@ -1015,7 +1015,7 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, cross_db.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary run: cross_kv compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.cross_populated = true; } @@ -1072,7 +1072,7 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, db.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary run: decoder prompt compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } // Per-sublayer dumps at layers {0, n_layers/2, n_layers-1}. @@ -1264,7 +1264,10 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, sb.graph); gs != GGML_STATUS_SUCCESS) { - break; + log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary run: decoder step compute failed (%d)", + static_cast(gs)); + commit_result(); + return TRANSCRIBE_ERR_BACKEND; } n_past += 1; @@ -1304,19 +1307,26 @@ transcribe_status run(transcribe_session * session, } if (!new_compute_ctx(4 * 1024 * 1024)) { - break; + log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary run: decoder step ggml_init failed"); + commit_result(); + return TRANSCRIBE_ERR_OOM; } DecoderBuild db_step = build_decoder_graph_kv(cc->compute_ctx, cm->weights, cm->hparams, cc->kv_cache, /*n_tokens=*/1, n_past, T_enc, /*skip_log_softmax=*/true, cc->decoder_use_flash); if (db_step.out == nullptr || db_step.graph == nullptr) { - break; + log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary run: decoder step graph build failed"); + commit_result(); + return TRANSCRIBE_ERR_GGUF; } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, db_step.graph)) { - break; + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, + "canary run: step graph allocation failed — out of memory."); + commit_result(); + return TRANSCRIBE_ERR_OOM; } int32_t token_id = next_token; @@ -1326,7 +1336,10 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, db_step.graph); gs != GGML_STATUS_SUCCESS) { - break; + log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary run: decoder step compute failed (%d)", + static_cast(gs)); + commit_result(); + return TRANSCRIBE_ERR_BACKEND; } n_past += 1; @@ -1418,7 +1431,7 @@ transcribe_status encode_one_to_host(CanarySession * cc, cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -1456,7 +1469,7 @@ transcribe_status encode_one_to_host(CanarySession * cc, const int64_t t0 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us += ggml_time_us() - t0; @@ -1701,7 +1714,7 @@ transcribe_status run_batch(transcribe_session * session, } ggml_backend_tensor_set(cross.encoder_out_in, packed.data(), 0, packed.size() * sizeof(float)); if (ggml_backend_sched_graph_compute(cc->sched, cross.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.cross_populated = true; } @@ -1730,18 +1743,18 @@ transcribe_status run_batch(transcribe_session * session, } StepBuildBatched sb{}; - auto rebuild = [&](int win, transcribe::EncDecStepIO & io) -> bool { + auto rebuild = [&](int win, transcribe::EncDecStepIO & io) -> transcribe_status { if (!new_compute_ctx(16 * 1024 * 1024)) { - return false; + return TRANSCRIBE_ERR_OOM; } sb = build_step_graph_batched(cc->compute_ctx, cm->weights, hp, cc->kv_cache, win, T_enc_max, n, cc->decoder_use_flash); if (sb.graph == nullptr || sb.argmax_out == nullptr) { - return false; + return TRANSCRIBE_ERR_GGUF; } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, sb.graph)) { - return false; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(sb.cross_mask_in, cmask.data(), 0, cmask.size() * sizeof(ggml_fp16_t)); io.token_ids = sb.token_ids_in; @@ -1750,7 +1763,7 @@ transcribe_status run_batch(transcribe_session * session, io.self_mask = sb.self_mask_in; io.argmax = sb.argmax_out; io.graph = sb.graph; - return true; + return TRANSCRIBE_OK; }; std::vector> generated(n); diff --git a/src/arch/canary_qwen/model.cpp b/src/arch/canary_qwen/model.cpp index 386fd9da..46a2c08d 100644 --- a/src/arch/canary_qwen/model.cpp +++ b/src/arch/canary_qwen/model.cpp @@ -246,7 +246,7 @@ transcribe_status fuse_batch_norm(CanaryQwenModel & m) { ggml_init_params params = { ctx_size, nullptr, true }; m.bn_fused_ctx = ggml_init(params); if (m.bn_fused_ctx == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } for (size_t i = 0; i < n_blocks; ++i) { @@ -257,7 +257,7 @@ transcribe_status fuse_batch_norm(CanaryQwenModel & m) { m.bn_fused_buffer = ggml_backend_alloc_ctx_tensors(m.bn_fused_ctx, m.plan.scheduler_list.back()); if (m.bn_fused_buffer == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } std::vector bn_w(d), bn_b(d), rm(d), rv(d); @@ -321,7 +321,7 @@ transcribe_status promote_conv_pw_to_f32_on_cpu(CanaryQwenModel & m) { ggml_init_params init_params = { ctx_size, nullptr, true }; m.conv_pw_f32_ctx = ggml_init(init_params); if (m.conv_pw_f32_ctx == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } std::vector replacements; @@ -331,7 +331,7 @@ transcribe_status promote_conv_pw_to_f32_on_cpu(CanaryQwenModel & m) { if (r == nullptr) { ggml_free(m.conv_pw_f32_ctx); m.conv_pw_f32_ctx = nullptr; - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } ggml_set_name(r, s.src->name); replacements.push_back(r); @@ -342,7 +342,7 @@ transcribe_status promote_conv_pw_to_f32_on_cpu(CanaryQwenModel & m) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary_qwen: conv F16->F32 promotion buffer alloc failed"); ggml_free(m.conv_pw_f32_ctx); m.conv_pw_f32_ctx = nullptr; - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } ggml_backend_buffer_set_usage(m.conv_pw_f32_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -447,7 +447,7 @@ transcribe_status promote_linears_bf16_to_f32_on_cpu(CanaryQwenModel & m) { ggml_init_params init_params = { ctx_size, nullptr, true }; m.linear_f32_ctx = ggml_init(init_params); if (m.linear_f32_ctx == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } std::vector replacements; @@ -457,7 +457,7 @@ transcribe_status promote_linears_bf16_to_f32_on_cpu(CanaryQwenModel & m) { if (r == nullptr) { ggml_free(m.linear_f32_ctx); m.linear_f32_ctx = nullptr; - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } ggml_set_name(r, s.src->name); replacements.push_back(r); @@ -468,7 +468,7 @@ transcribe_status promote_linears_bf16_to_f32_on_cpu(CanaryQwenModel & m) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary_qwen: BF16→F32 linear promotion buffer alloc failed"); ggml_free(m.linear_f32_ctx); m.linear_f32_ctx = nullptr; - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } ggml_backend_buffer_set_usage(m.linear_f32_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -650,7 +650,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary_qwen: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -680,10 +680,12 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par for (auto & b : m->weights.dec_blocks) { entries.push_back({ b.ffn_gate_w, b.ffn_up_w, &b.ffn_gate_up_w }); } - if (!transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, - entries, m->packed_gate_up, "canary_qwen")) { + if (const transcribe_status st = + transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, + entries, m->packed_gate_up, "canary_qwen"); + st != TRANSCRIBE_OK) { m->packed_gate_up.free(); - return TRANSCRIBE_ERR_GGUF; + return st; } } @@ -815,7 +817,7 @@ transcribe_status run(transcribe_session * context, cc->compute_ctx = ggml_init(ip); if (cc->compute_ctx == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary_qwen run: ggml_init failed (encoder)"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } } @@ -831,7 +833,7 @@ transcribe_status run(transcribe_session * context, /*op_offload=*/true); if (cc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary_qwen run: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -865,7 +867,7 @@ transcribe_status run(transcribe_session * context, const int64_t t_enc_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary_qwen run: encoder compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_encode_us = ggml_time_us() - t_enc_start; @@ -1028,7 +1030,7 @@ transcribe_status run(transcribe_session * context, const int64_t t0 = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, pb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary_qwen run: prefill compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } t_prefill_compute_us = ggml_time_us() - t0; t_prefill_us = t_prefill_compute_us; @@ -1175,7 +1177,7 @@ transcribe_status run(transcribe_session * context, const int64_t t_c0 = profile_decode ? ggml_time_us() : 0; if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, sb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "canary_qwen run: step compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } if (profile_decode) { const int64_t dt = ggml_time_us() - t_c0; @@ -1287,7 +1289,7 @@ transcribe_status reset_ctx(CanaryQwenSession * cc, int mb) { ip.mem_buffer = nullptr; ip.no_alloc = true; cc->compute_ctx = ggml_init(ip); - return cc->compute_ctx != nullptr ? TRANSCRIBE_OK : TRANSCRIBE_ERR_GGUF; + return cc->compute_ctx != nullptr ? TRANSCRIBE_OK : TRANSCRIBE_ERR_OOM; } // Conformer encoder (+ perception) for one utterance from a PRECOMPUTED mel @@ -1305,7 +1307,7 @@ transcribe_status encode_one(CanaryQwenSession * cc, } if (reset_ctx(cc, 32) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } EncoderBuild eb = build_encoder_graph(cc->compute_ctx, cm->weights, hp, mel_n_frames, /*kv_type=*/GGML_TYPE_COUNT, cc->encoder_use_flash, cm->backend.c_str()); @@ -1317,12 +1319,12 @@ transcribe_status encode_one(CanaryQwenSession * cc, cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, eb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(eb.mel_in, mel_buf.data(), 0, mel_buf.size() * sizeof(float)); @@ -1337,7 +1339,7 @@ transcribe_status encode_one(CanaryQwenSession * cc, const int64_t t_enc0 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us += ggml_time_us() - t_enc0; @@ -1517,7 +1519,7 @@ transcribe_status run_batch(transcribe_session * session, std::vector> generated(n); { if (reset_ctx(cc, 32) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } PrefillBuildBatched pb = build_prefill_graph_batched(cc->compute_ctx, cm->weights, hp, cc->kv_cache_batch, max_T_prompt, T_enc_max, n, cc->decoder_use_flash); @@ -1526,7 +1528,7 @@ transcribe_status run_batch(transcribe_session * session, } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, pb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } const int hidden = hp.dec_hidden; @@ -1583,7 +1585,7 @@ transcribe_status run_batch(transcribe_session * session, ggml_backend_tensor_set(pb.last_idx_in, lidx.data(), 0, lidx.size() * sizeof(int32_t)); apply_thread_policy(cc); if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } std::vector amax(n, 0); ggml_backend_tensor_get(pb.out, amax.data(), 0, amax.size() * sizeof(int32_t)); @@ -1601,7 +1603,7 @@ transcribe_status run_batch(transcribe_session * session, const int32_t eos_id = cm->hparams.eos_token_id; if (reset_ctx(cc, 16) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } StepBuildBatched sb = build_step_graph_batched(cc->compute_ctx, cm->weights, hp, cc->kv_cache_batch, max_n_kv, n, cc->decoder_use_flash); @@ -1610,7 +1612,7 @@ transcribe_status run_batch(transcribe_session * session, } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, sb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } transcribe::causal_lm::StepBatchedIO io{}; diff --git a/src/arch/cohere/model.cpp b/src/arch/cohere/model.cpp index bdf8050e..7d593af5 100644 --- a/src/arch/cohere/model.cpp +++ b/src/arch/cohere/model.cpp @@ -297,7 +297,7 @@ transcribe_status fuse_batch_norm(CohereModel & m) { ggml_init_params params = { ctx_size, nullptr, true }; m.bn_fused_ctx = ggml_init(params); if (m.bn_fused_ctx == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } for (size_t i = 0; i < n_blocks; ++i) { @@ -308,7 +308,7 @@ transcribe_status fuse_batch_norm(CohereModel & m) { m.bn_fused_buffer = ggml_backend_alloc_ctx_tensors(m.bn_fused_ctx, m.plan.scheduler_list.back()); if (m.bn_fused_buffer == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } std::vector bn_w(d), bn_b(d), rm(d), rv(d); @@ -574,7 +574,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "cohere: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -736,7 +736,7 @@ transcribe_status run(transcribe_session * session, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "cohere run: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -787,7 +787,7 @@ transcribe_status run(transcribe_session * session, const int64_t t_enc_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "cohere run: encoder graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_encode_us = ggml_time_us() - t_enc_start; @@ -959,7 +959,7 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, cross_db.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "cohere run: cross_kv compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.cross_populated = true; } @@ -1015,7 +1015,7 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, db.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "cohere run: decoder prompt compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } // Dump decoder intermediates (prompt pass only). @@ -1204,7 +1204,7 @@ transcribe_status run(transcribe_session * session, log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "cohere run: step compute failed (%d, n_past=%d)", static_cast(gs), n_past); commit_result(); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } n_past += 1; @@ -1253,19 +1253,26 @@ transcribe_status run(transcribe_session * session, } if (!new_compute_ctx(4 * 1024 * 1024)) { - break; + log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "cohere run: decoder step ggml_init failed"); + commit_result(); + return TRANSCRIBE_ERR_OOM; } DecoderBuild db_step = build_decoder_graph_kv(cc->compute_ctx, cm->weights, cm->hparams, cc->kv_cache, /*n_tokens=*/1, n_past, T_enc, /*skip_log_softmax=*/true, cc->decoder_use_flash); if (db_step.out == nullptr || db_step.graph == nullptr) { - break; + log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "cohere run: decoder step graph build failed"); + commit_result(); + return TRANSCRIBE_ERR_GGUF; } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, db_step.graph)) { - break; + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, + "cohere run: step graph allocation failed — out of memory."); + commit_result(); + return TRANSCRIBE_ERR_OOM; } int32_t token_id = next_token; @@ -1275,7 +1282,10 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, db_step.graph); gs != GGML_STATUS_SUCCESS) { - break; + log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "cohere run: decoder step compute failed (%d)", + static_cast(gs)); + commit_result(); + return TRANSCRIBE_ERR_BACKEND; } n_past += 1; @@ -1379,7 +1389,7 @@ transcribe_status encode_one_to_host(CohereSession * cc, cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -1416,7 +1426,7 @@ transcribe_status encode_one_to_host(CohereSession * cc, const int64_t t0 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us += ggml_time_us() - t0; @@ -1645,7 +1655,7 @@ transcribe_status run_batch(transcribe_session * session, } ggml_backend_tensor_set(cross.encoder_out_in, packed.data(), 0, packed.size() * sizeof(float)); if (ggml_backend_sched_graph_compute(cc->sched, cross.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.cross_populated = true; } @@ -1676,18 +1686,18 @@ transcribe_status run_batch(transcribe_session * session, } StepBuildBatched sb{}; - auto rebuild = [&](int win, transcribe::EncDecStepIO & io) -> bool { + auto rebuild = [&](int win, transcribe::EncDecStepIO & io) -> transcribe_status { if (!new_compute_ctx(16 * 1024 * 1024)) { - return false; + return TRANSCRIBE_ERR_OOM; } sb = build_step_graph_batched(cc->compute_ctx, cm->weights, hp, cc->kv_cache, win, T_enc_max, n, cc->decoder_use_flash); if (sb.graph == nullptr || sb.argmax_out == nullptr) { - return false; + return TRANSCRIBE_ERR_GGUF; } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, sb.graph)) { - return false; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(sb.cross_mask_in, cmask.data(), 0, cmask.size() * sizeof(ggml_fp16_t)); io.token_ids = sb.token_ids_in; @@ -1696,7 +1706,7 @@ transcribe_status run_batch(transcribe_session * session, io.self_mask = sb.self_mask_in; io.argmax = sb.argmax_out; io.graph = sb.graph; - return true; + return TRANSCRIBE_OK; }; std::vector> generated(n); diff --git a/src/arch/funasr_nano/model.cpp b/src/arch/funasr_nano/model.cpp index 751d6ede..301955d0 100644 --- a/src/arch/funasr_nano/model.cpp +++ b/src/arch/funasr_nano/model.cpp @@ -360,7 +360,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "funasr_nano: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -380,10 +380,12 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par for (auto & b : m->weights.dec_blocks) { entries.push_back({ b.ffn_gate_w, b.ffn_up_w, &b.ffn_gate_up_w }); } - if (!transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, - entries, m->packed_gate_up, "funasr_nano")) { + if (const transcribe_status st = + transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, + entries, m->packed_gate_up, "funasr_nano"); + st != TRANSCRIBE_OK) { m->packed_gate_up.free(); - return TRANSCRIBE_ERR_GGUF; + return st; } } @@ -508,7 +510,7 @@ transcribe_status run(transcribe_session * session, cc->compute_ctx = ggml_init(ip); if (cc->compute_ctx == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "funasr_nano run: ggml_init for compute_ctx failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } } @@ -524,7 +526,7 @@ transcribe_status run(transcribe_session * session, /*op_offload=*/true); if (cc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "funasr_nano run: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -544,7 +546,7 @@ transcribe_status run(transcribe_session * session, const int64_t t_enc_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "funasr_nano run: encoder graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_encode_us = ggml_time_us() - t_enc_start; @@ -596,6 +598,9 @@ transcribe_status run(transcribe_session * session, ip.mem_buffer = nullptr; ip.no_alloc = true; cc->compute_ctx = ggml_init(ip); + if (cc->compute_ctx == nullptr) { + return TRANSCRIBE_ERR_OOM; + } } AdaptorBuild ab = build_adaptor_graph(cc->compute_ctx, cm->weights, hp, T_lfr); @@ -612,7 +617,7 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, ab.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "funasr_nano run: adaptor graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } try_dump("adaptor.linear1.out", ab.dumps.linear1_out, "adaptor.linear1"); @@ -783,7 +788,7 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, pb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "funasr_nano run: prefill graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.n = T_prompt; @@ -845,6 +850,9 @@ transcribe_status run(transcribe_session * session, ip.mem_buffer = nullptr; ip.no_alloc = true; cc->compute_ctx = ggml_init(ip); + if (cc->compute_ctx == nullptr) { + return TRANSCRIBE_ERR_OOM; + } } StepBuild sb = build_step_graph(cc->compute_ctx, cm->weights, hp, cc->kv_cache, max_n_kv, cc->decoder_use_flash); if (sb.graph == nullptr || sb.out == nullptr) { @@ -883,7 +891,7 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, sb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "funasr_nano run: step graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } int32_t argmax_tok = 0; @@ -960,7 +968,7 @@ transcribe_status reset_ctx(FunAsrNanoSession * cc, int mb) { ip.mem_buffer = nullptr; ip.no_alloc = true; cc->compute_ctx = ggml_init(ip); - return cc->compute_ctx != nullptr ? TRANSCRIBE_OK : TRANSCRIBE_ERR_GGUF; + return cc->compute_ctx != nullptr ? TRANSCRIBE_OK : TRANSCRIBE_ERR_OOM; } // encoder + adaptor for one utterance from a PRECOMPUTED frontend buffer @@ -990,12 +998,12 @@ transcribe_status audio_embed_one(FunAsrNanoSession * cc, cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, eb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(eb.frontend_in, frontend_buf.data(), 0, frontend_buf.size() * sizeof(float)); transcribe::sanm::build_sinusoidal_pe(cc->pe_buf, hp.enc_d_input, T_lfr); @@ -1003,7 +1011,7 @@ transcribe_status audio_embed_one(FunAsrNanoSession * cc, apply_thread_policy(cc); const int64_t t_enc0 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us += ggml_time_us() - t_enc0; cc->enc_host.resize(static_cast(hp.enc_d_model) * T_lfr); @@ -1019,12 +1027,12 @@ transcribe_status audio_embed_one(FunAsrNanoSession * cc, } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, ab.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(ab.enc_in, cc->enc_host.data(), 0, cc->enc_host.size() * sizeof(float)); const int64_t t_enc1 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, ab.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us += ggml_time_us() - t_enc1; cc->adaptor_host.resize(static_cast(hp.adaptor_llm_dim) * T_lfr); @@ -1221,7 +1229,7 @@ transcribe_status run_batch(transcribe_session * session, } T_audio_max = std::max(1, T_audio_max); if (reset_ctx(cc, 32) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } PrefillBuildBatched pb = build_prefill_graph_batched(cc->compute_ctx, cm->weights, hp, cc->kv_cache_batch, max_T_prompt, T_audio_max, n, cc->decoder_use_flash); @@ -1230,7 +1238,7 @@ transcribe_status run_batch(transcribe_session * session, } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, pb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } const int hidden = hp.dec_hidden; @@ -1287,7 +1295,7 @@ transcribe_status run_batch(transcribe_session * session, ggml_backend_tensor_set(pb.last_idx_in, lidx.data(), 0, lidx.size() * sizeof(int32_t)); apply_thread_policy(cc); if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } std::vector amax(n, 0); ggml_backend_tensor_get(pb.out, amax.data(), 0, amax.size() * sizeof(int32_t)); @@ -1305,7 +1313,7 @@ transcribe_status run_batch(transcribe_session * session, const int32_t eos_id = cm->hparams.eos_token_id; if (reset_ctx(cc, 16) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } StepBuildBatched sb = build_step_graph_batched(cc->compute_ctx, cm->weights, hp, cc->kv_cache_batch, max_n_kv, n, cc->decoder_use_flash); @@ -1314,7 +1322,7 @@ transcribe_status run_batch(transcribe_session * session, } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, sb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } transcribe::causal_lm::StepBatchedIO io{}; diff --git a/src/arch/gigaam/model.cpp b/src/arch/gigaam/model.cpp index 6fe2c787..8d036b1b 100644 --- a/src/arch/gigaam/model.cpp +++ b/src/arch/gigaam/model.cpp @@ -143,7 +143,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "gigaam: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -408,11 +408,8 @@ transcribe_status run(transcribe_session * session, const float * pcm, int n_sam static_cast(gm->plan.scheduler_list.size()), /*graph_size=*/8192, /*parallel=*/false, /*op_offload=*/true); if (gc->sched == nullptr) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, - "gigaam run: scheduler allocation failed — out of memory. " - "Split long audio into shorter segments (see " - "transcribe_capabilities.max_audio_ms)."); - return TRANSCRIBE_ERR_OOM; + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "gigaam run: ggml_backend_sched_new failed"); + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(gc->sched); @@ -441,7 +438,7 @@ transcribe_status run(transcribe_session * session, const float * pcm, int n_sam const int64_t t_enc_start = ggml_time_us(); if (ggml_backend_sched_graph_compute(gc->sched, eb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "gigaam: graph_compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } gc->t_encode_us = ggml_time_us() - t_enc_start; @@ -569,11 +566,8 @@ transcribe_status run_batch_encode(GigaamSession * gc, static_cast(gm->plan.scheduler_list.size()), /*graph_size=*/8192, /*parallel=*/false, /*op_offload=*/true); if (gc->sched == nullptr) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, - "gigaam run: scheduler allocation failed — out of memory. " - "Split long audio into shorter segments (see " - "transcribe_capabilities.max_audio_ms)."); - return TRANSCRIBE_ERR_OOM; + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "gigaam run: ggml_backend_sched_new failed"); + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(gc->sched); @@ -651,7 +645,7 @@ transcribe_status run_batch_encode(GigaamSession * gc, const int64_t t_enc_start = ggml_time_us(); if (ggml_backend_sched_graph_compute(gc->sched, eb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "gigaam run_batch: graph_compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } gc->t_encode_us = ggml_time_us() - t_enc_start; diff --git a/src/arch/granite/model.cpp b/src/arch/granite/model.cpp index 959c314a..f51d195c 100644 --- a/src/arch/granite/model.cpp +++ b/src/arch/granite/model.cpp @@ -401,7 +401,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -430,10 +430,12 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par for (auto & b : m->weights.dec_blocks) { entries.push_back({ b.ffn_gate_w, b.ffn_up_w, &b.ffn_gate_up_w }); } - if (!transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, - entries, m->packed_gate_up, "granite")) { + if (const transcribe_status st = + transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, + entries, m->packed_gate_up, "granite"); + st != TRANSCRIBE_OK) { m->packed_gate_up.free(); - return TRANSCRIBE_ERR_GGUF; + return st; } } @@ -841,7 +843,7 @@ transcribe_status run(transcribe_session * ctx_base, static_cast(cm->plan.scheduler_list.size()), 32768, false, true); if (cc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite run: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -891,7 +893,7 @@ transcribe_status run(transcribe_session * ctx_base, const int64_t t_enc_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite run: encoder graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_encode_us = ggml_time_us() - t_enc_start; @@ -969,7 +971,7 @@ transcribe_status run(transcribe_session * ctx_base, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, pb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite run: projector graph compute failed (%d)", static_cast(gs)); ggml_free(proj_ctx); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } try_dump("proj.qformer.out", pb.dumps.qformer_out, "projector"); @@ -1153,7 +1155,7 @@ transcribe_status run(transcribe_session * ctx_base, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, dec.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite run: decoder prefill compute failed (%d)", static_cast(gs)); ggml_free(dec_ctx); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_decode_us = ggml_time_us() - t_dec_start; @@ -1211,6 +1213,9 @@ transcribe_status run(transcribe_session * ctx_base, ip.mem_buffer = nullptr; ip.no_alloc = true; step_ctx = ggml_init(ip); + if (step_ctx == nullptr) { + return TRANSCRIBE_ERR_OOM; + } } StepBuild step = build_step_graph(step_ctx, cm->weights, cm->hparams, cc->kv, max_n_kv, cc->decoder_use_flash); if (step.graph == nullptr) { @@ -1259,7 +1264,7 @@ transcribe_status run(transcribe_session * ctx_base, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, step.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite run: step compute failed (%d)", static_cast(gs)); ggml_free(step_ctx); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } int32_t amax = 0; @@ -1316,7 +1321,7 @@ transcribe_status reset_ctx_g(GraniteSession * cc, int mb) { ip.mem_buffer = nullptr; ip.no_alloc = true; cc->compute_ctx = ggml_init(ip); - return cc->compute_ctx != nullptr ? TRANSCRIBE_OK : TRANSCRIBE_ERR_GGUF; + return cc->compute_ctx != nullptr ? TRANSCRIBE_OK : TRANSCRIBE_ERR_OOM; } void apply_threads_g(GraniteSession * cc) { @@ -1355,7 +1360,7 @@ transcribe_status encode_one(GraniteSession * cc, cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 32768, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -1389,7 +1394,7 @@ transcribe_status encode_one(GraniteSession * cc, const int64_t t_enc0 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us += ggml_time_us() - t_enc0; @@ -1433,7 +1438,7 @@ transcribe_status encode_one(GraniteSession * cc, const int64_t t_enc1 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { ggml_free(proj_ctx); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us += ggml_time_us() - t_enc1; n_audio_out = pb.n_audio_tokens; @@ -1714,7 +1719,7 @@ transcribe_status run_batch(transcribe_session * session, ggml_backend_tensor_set(pb.last_idx_in, lidx.data(), 0, lidx.size() * sizeof(int32_t)); apply_threads_g(cc); if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } std::vector amax(n, 0); ggml_backend_tensor_get(pb.out, amax.data(), 0, amax.size() * sizeof(int32_t)); diff --git a/src/arch/granite5_ctc/model.cpp b/src/arch/granite5_ctc/model.cpp index 2d80d597..0e00d644 100644 --- a/src/arch/granite5_ctc/model.cpp +++ b/src/arch/granite5_ctc/model.cpp @@ -223,7 +223,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite5_ctc: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -302,10 +302,8 @@ transcribe_status run_encoder(Granite5CtcSession * cc, static_cast(cm->plan.scheduler_list.size()), /*graph_size=*/16384, /*parallel=*/false, /*op_offload=*/true); if (cc->sched == nullptr) { - log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, - "granite5_ctc: scheduler allocation failed — out of memory. " - "Split long audio into shorter segments, or use a smaller batch."); - return TRANSCRIBE_ERR_OOM; + log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite5_ctc: ggml_backend_sched_new failed"); + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -353,7 +351,7 @@ transcribe_status run_encoder(Granite5CtcSession * cc, const int64_t t_enc_start = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, out_eb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite5_ctc: graph_compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_encode_us = ggml_time_us() - t_enc_start; diff --git a/src/arch/granite_nar/model.cpp b/src/arch/granite_nar/model.cpp index ed6d2d30..c90b3c9b 100644 --- a/src/arch/granite_nar/model.cpp +++ b/src/arch/granite_nar/model.cpp @@ -343,7 +343,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite_nar: alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -460,7 +460,7 @@ transcribe_status run(transcribe_session * ctx_base, static_cast(cm->plan.scheduler_list.size()), 32768, false, true); if (cc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite_nar run: sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -498,7 +498,7 @@ transcribe_status run(transcribe_session * ctx_base, const int64_t t_enc_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite_nar run: encoder compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_encode_us = ggml_time_us() - t_enc_start; @@ -623,7 +623,7 @@ transcribe_status run(transcribe_session * ctx_base, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, pb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite_nar run: projector compute failed (%d)", static_cast(gs)); ggml_free(proj_ctx); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } try_dump("proj.qformer.out", pb.dumps.qformer_out, "projector"); @@ -724,7 +724,7 @@ transcribe_status run(transcribe_session * ctx_base, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, fb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "granite_nar run: decoder compute failed (%d)", static_cast(gs)); ggml_free(dec_ctx); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_decode_us = ggml_time_us() - t_dec_start; diff --git a/src/arch/medasr/model.cpp b/src/arch/medasr/model.cpp index 46fae6be..c1f33e9a 100644 --- a/src/arch/medasr/model.cpp +++ b/src/arch/medasr/model.cpp @@ -191,7 +191,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "medasr: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -479,11 +479,8 @@ transcribe_status run(transcribe_session * session, const float * pcm, int n_sam static_cast(gm->plan.scheduler_list.size()), /*graph_size=*/8192, /*parallel=*/false, /*op_offload=*/true); if (gc->sched == nullptr) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, - "medasr run: scheduler allocation failed — out of memory. " - "Split long audio into shorter segments (see " - "transcribe_capabilities.max_audio_ms)."); - return TRANSCRIBE_ERR_OOM; + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "medasr run: ggml_backend_sched_new failed"); + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(gc->sched); @@ -527,7 +524,7 @@ transcribe_status run(transcribe_session * session, const float * pcm, int n_sam const int64_t t_enc_start = ggml_time_us(); if (ggml_backend_sched_graph_compute(gc->sched, eb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "medasr: graph_compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } gc->t_encode_us = ggml_time_us() - t_enc_start; @@ -752,11 +749,8 @@ transcribe_status run_batch_encode(MedAsrSession * gc, static_cast(gm->plan.scheduler_list.size()), /*graph_size=*/8192, /*parallel=*/false, /*op_offload=*/true); if (gc->sched == nullptr) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, - "medasr run: scheduler allocation failed — out of memory. " - "Split long audio into shorter segments (see " - "transcribe_capabilities.max_audio_ms)."); - return TRANSCRIBE_ERR_OOM; + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "medasr run: ggml_backend_sched_new failed"); + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(gc->sched); @@ -846,7 +840,7 @@ transcribe_status run_batch_encode(MedAsrSession * gc, const int64_t t_enc_start = ggml_time_us(); if (ggml_backend_sched_graph_compute(gc->sched, eb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "medasr run_batch: graph_compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } gc->t_encode_us = ggml_time_us() - t_enc_start; diff --git a/src/arch/moonshine/model.cpp b/src/arch/moonshine/model.cpp index 9f3de4a4..22e4bebc 100644 --- a/src/arch/moonshine/model.cpp +++ b/src/arch/moonshine/model.cpp @@ -258,7 +258,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -385,7 +385,7 @@ transcribe_status run(transcribe_session * session, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine run: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -408,7 +408,7 @@ transcribe_status run(transcribe_session * session, if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine run: encoder compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } auto try_dump = [](const char * name, ggml_tensor * t, const char * stage) { @@ -481,7 +481,7 @@ transcribe_status run(transcribe_session * session, ggml_backend_tensor_set(cross_db.encoder_out_in, cc->enc_host.data(), 0, cc->enc_host.size() * sizeof(float)); if (ggml_backend_sched_graph_compute(cc->sched, cross_db.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine run: cross_kv compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.cross_populated = true; } @@ -550,7 +550,7 @@ transcribe_status run(transcribe_session * session, if (ggml_backend_sched_graph_compute(cc->sched, db.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine run: decoder compute failed (n_past=%d)", n_past_in); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } if (dump_prompt) { @@ -695,7 +695,7 @@ transcribe_status run(transcribe_session * session, if (ggml_backend_sched_graph_compute(cc->sched, sb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine run: step compute failed (n_past=%d)", n_past); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } n_past += 1; @@ -818,7 +818,7 @@ transcribe_status encode_one_to_host(MoonshineSession * cc, cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -840,7 +840,7 @@ transcribe_status encode_one_to_host(MoonshineSession * cc, const int64_t t0 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us += ggml_time_us() - t0; @@ -995,7 +995,7 @@ transcribe_status run_batch(transcribe_session * session, } ggml_backend_tensor_set(cross.encoder_out_in, packed.data(), 0, packed.size() * sizeof(float)); if (ggml_backend_sched_graph_compute(cc->sched, cross.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.cross_populated = true; } @@ -1014,24 +1014,24 @@ transcribe_status run_batch(transcribe_session * session, } StepBuildBatched sb{}; - auto rebuild = [&](int win, transcribe::EncDecStepIO & io) -> bool { + auto rebuild = [&](int win, transcribe::EncDecStepIO & io) -> transcribe_status { if (!ensure_compute_ctx(cc, 16 * 1024 * 1024)) { transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine run_batch: compute context allocation failed " "(step) — out of memory."); - return false; + return TRANSCRIBE_ERR_OOM; } sb = build_step_graph_batched(cc->compute_ctx, cm->weights, hp, cc->kv_cache, win, T_enc_max, n, /*use_flash=*/true); if (sb.graph == nullptr || sb.argmax_out == nullptr) { - return false; + return TRANSCRIBE_ERR_GGUF; } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, sb.graph)) { transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine run_batch: step graph allocation failed — " "out of memory."); - return false; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(sb.cross_mask_in, cmask.data(), 0, cmask.size() * sizeof(ggml_fp16_t)); io.token_ids = sb.token_ids_in; @@ -1040,7 +1040,7 @@ transcribe_status run_batch(transcribe_session * session, io.self_mask = sb.self_mask_in; io.argmax = sb.argmax_out; io.graph = sb.graph; - return true; + return TRANSCRIBE_OK; }; const std::vector prompt_ids = { static_cast(decoder_start) }; diff --git a/src/arch/moonshine_streaming/model.cpp b/src/arch/moonshine_streaming/model.cpp index 524dd31a..68626064 100644 --- a/src/arch/moonshine_streaming/model.cpp +++ b/src/arch/moonshine_streaming/model.cpp @@ -271,7 +271,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine_streaming: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -364,7 +364,7 @@ transcribe_status ensure_sched(MoonshineStreamingSession * cc, MoonshineStreamin static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine_streaming: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } return TRANSCRIBE_OK; } @@ -503,7 +503,7 @@ transcribe_status encode_window_to_host(MoonshineStreamingSession * cc, if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine_streaming encode_window: encoder compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } if (emit_dumps) { @@ -598,7 +598,7 @@ transcribe_status apply_adapter_window(MoonshineStreamingSession * cc, if (ggml_backend_sched_graph_compute(cc->sched, ab.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine_streaming adapter: compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } if (emit_dumps) { @@ -681,7 +681,7 @@ transcribe_status project_cross_kv_window(MoonshineStreamingSession * cc, if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine_streaming cross_kv_proj: compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } const size_t per_slice_floats = static_cast(dec_h) * static_cast(n_frames); @@ -807,7 +807,7 @@ transcribe_status commit_cross_kv_from_host(MoonshineStreamingSession * if (ggml_backend_sched_graph_compute(cc->sched, cb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine_streaming cross_kv_commit: compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.cross_populated = true; @@ -914,7 +914,7 @@ transcribe_status decode_from_kv_cache(MoonshineStreamingSession * cc, if (ggml_backend_sched_graph_compute(cc->sched, db.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine_streaming decode: decoder compute failed (n_past=%d)", n_past_in); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } if (dump_prompt) { @@ -1119,7 +1119,7 @@ transcribe_status decode_from_committed_enc(MoonshineStreamingSession * cc, ggml_backend_tensor_set(cross_db.encoder_out_in, adapter_host.data(), 0, adapter_host.size() * sizeof(float)); if (ggml_backend_sched_graph_compute(cc->sched, cross_db.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine_streaming decode: cross_kv compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.cross_populated = true; cc->t_decode_us += ggml_time_us() - t_xkv_start; @@ -1950,7 +1950,7 @@ transcribe_status run_batch(transcribe_session * session, } ggml_backend_tensor_set(cross.encoder_out_in, packed.data(), 0, packed.size() * sizeof(float)); if (ggml_backend_sched_graph_compute(cc->sched, cross.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.cross_populated = true; } @@ -1981,30 +1981,30 @@ transcribe_status run_batch(transcribe_session * session, } StepBuildBatched sb{}; - auto rebuild_step = [&](int win) -> bool { + auto rebuild_step = [&](int win) -> transcribe_status { if (!ensure_compute_ctx(cc, 16 * 1024 * 1024)) { transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine_streaming run_batch: compute context allocation " "failed (step) — out of memory."); - return false; + return TRANSCRIBE_ERR_OOM; } sb = build_step_graph_batched(cc->compute_ctx, cm->weights, hp, cc->kv_cache, win, T_enc_max, n, /*use_flash=*/true); if (sb.graph == nullptr || sb.argmax_out == nullptr) { - return false; + return TRANSCRIBE_ERR_GGUF; } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, sb.graph)) { transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moonshine_streaming run_batch: step graph allocation failed — " "out of memory."); - return false; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(sb.cross_mask_in, cmask.data(), 0, cmask.size() * sizeof(ggml_fp16_t)); - return true; + return TRANSCRIBE_OK; }; - if (!rebuild_step(kv_window)) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = rebuild_step(kv_window); st != TRANSCRIBE_OK) { + return st; } std::vector smask(static_cast(kv_window) * n, f16_ninf); @@ -2030,16 +2030,16 @@ transcribe_status run_batch(transcribe_session * session, ggml_backend_tensor_set(sb.kv_idx_in, kvidx_buf.data(), 0, n * sizeof(int64_t)); ggml_backend_tensor_set(sb.self_mask_in, smask.data(), 0, smask.size() * sizeof(ggml_fp16_t)); if (ggml_backend_sched_graph_compute(cc->sched, sb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } ggml_backend_tensor_get(sb.argmax_out, argmax_buf.data(), 0, n * sizeof(int32_t)); return TRANSCRIBE_OK; }; // Grow the window (and rebuild the graph + mask) so position `pos` fits. - auto ensure_window = [&](int pos) -> bool { + auto ensure_window = [&](int pos) -> transcribe_status { if (pos + 1 <= kv_window) { - return true; + return TRANSCRIBE_OK; } int win = kv_window; while (win < pos + 1 && win < n_ctx_cap) { @@ -2049,7 +2049,7 @@ transcribe_status run_batch(transcribe_session * session, win = n_ctx_cap; } if (win == kv_window) { - return true; + return TRANSCRIBE_OK; } // Re-fill a wider mask: positions [0, pos) already written for all b. std::vector wider(static_cast(win) * n, f16_ninf); @@ -2066,8 +2066,8 @@ transcribe_status run_batch(transcribe_session * session, for (int b = 0; b < n; ++b) { tok_buf[b] = decoder_start; } - if (run_step(0) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = run_step(0); st != TRANSCRIBE_OK) { + return st; } for (int b = 0; b < n; ++b) { if (finished[b]) { @@ -2094,14 +2094,14 @@ transcribe_status run_batch(transcribe_session * session, if (all_done || pos + 1 > n_ctx_cap) { break; } - if (!ensure_window(pos)) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = ensure_window(pos); st != TRANSCRIBE_OK) { + return st; } for (int b = 0; b < n; ++b) { tok_buf[b] = finished[b] ? eos : next_tok[b]; } - if (run_step(pos) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = run_step(pos); st != TRANSCRIBE_OK) { + return st; } for (int b = 0; b < n; ++b) { if (finished[b]) { diff --git a/src/arch/moss/model.cpp b/src/arch/moss/model.cpp index 5133a957..fb67560e 100644 --- a/src/arch/moss/model.cpp +++ b/src/arch/moss/model.cpp @@ -318,7 +318,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moss: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -337,10 +337,12 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par for (auto & b : m->weights.dec_blocks) { entries.push_back({ b.ffn_gate_w, b.ffn_up_w, &b.ffn_gate_up_w }); } - if (!transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, - entries, m->packed_gate_up, "moss")) { + if (const transcribe_status st = + transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, + entries, m->packed_gate_up, "moss"); + st != TRANSCRIBE_OK) { m->packed_gate_up.free(); - return TRANSCRIBE_ERR_GGUF; + return st; } } @@ -389,7 +391,7 @@ transcribe_status ensure_sched(MossSession * cc, MossModel * cm) { cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } return TRANSCRIBE_OK; @@ -405,7 +407,7 @@ transcribe_status reset_compute_ctx(MossSession * cc, int mb) { ip.mem_buffer = nullptr; ip.no_alloc = true; cc->compute_ctx = ggml_init(ip); - return cc->compute_ctx != nullptr ? TRANSCRIBE_OK : TRANSCRIBE_ERR_GGUF; + return cc->compute_ctx != nullptr ? TRANSCRIBE_OK : TRANSCRIBE_ERR_OOM; } // Fills enc_out [dec_hidden, T_enc]; returns T_enc via out_T_enc. `dumps` marks @@ -477,7 +479,7 @@ transcribe_status encode_one(MossSession * cc, const int64_t t_enc0 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moss encode: encoder graph compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us += ggml_time_us() - t_enc0; @@ -536,7 +538,7 @@ transcribe_status encode_one(MossSession * cc, const int64_t t_enc1 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, ab.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moss encode: adaptor graph compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us += ggml_time_us() - t_enc1; @@ -626,7 +628,7 @@ transcribe_status prefill_chunked(MossSession * cc, if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moss prefill: chunk %d/%d compute failed", c + 1, n_chunks); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.n = max_n_kv; @@ -847,7 +849,7 @@ transcribe_status run(transcribe_session * session, const int64_t t_pf0 = perf_debug ? ggml_time_us() : 0; if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moss run: prefill compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } t_prefill_us = perf_debug ? (ggml_time_us() - t_pf0) : 0; cc->kv_cache.n = T_prompt; @@ -940,7 +942,7 @@ transcribe_status run(transcribe_session * session, if (ggml_backend_sched_graph_compute(cc->sched, sb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "moss step: graph compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } if (perf_debug) { const int64_t dt = ggml_time_us() - t_c0; @@ -1281,7 +1283,7 @@ transcribe_status run_batch(transcribe_session * session, transcribe::configure_sched_n_threads(cc->sched, cc->n_threads); if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } std::vector amax(n, 0); ggml_backend_tensor_get(pb.out, amax.data(), 0, amax.size() * sizeof(int32_t)); diff --git a/src/arch/parakeet/decoder.cpp b/src/arch/parakeet/decoder.cpp index 27a83c2f..c52f295a 100644 --- a/src/arch/parakeet/decoder.cpp +++ b/src/arch/parakeet/decoder.cpp @@ -797,7 +797,8 @@ namespace { // step's (h, c) into new_state, returns a borrowed pointer into // new_state.h.back(). Feeds prev state + embedding into the resident // per-call graph, computes, reads new state back into the host LstmState. -// Caller guarantees g.ready and that new_state is sized to (L, H). +// Caller guarantees g.ready and that new_state is sized to (L, H). Returns +// nullptr if the graph compute fails. const float * predictor_step_ggml(const HostPredictor & predictor, PredGraph & g, int last_token, @@ -824,7 +825,10 @@ const float * predictor_step_ggml(const HostPredictor & predictor, ggml_backend_tensor_set(g.pc[l], prev_state.c[static_cast(l)].data(), 0, hb); } - ggml_backend_graph_compute(g.backend, g.graph); + if (const ggml_status gs = ggml_backend_graph_compute(g.backend, g.graph); gs != GGML_STATUS_SUCCESS) { + log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet decoder: predictor compute failed (%d)", static_cast(gs)); + return nullptr; + } for (int l = 0; l < g.L; ++l) { ggml_backend_tensor_get(g.nh[l], new_state.h[static_cast(l)].data(), 0, hb); @@ -843,8 +847,9 @@ const float * predictor_step_ggml(const HostPredictor & predictor, // summed = enc_proj + pred_proj [joint_h] // activated = activation(summed) [joint_h] // logits = out_w @ activated + out_b [joint_n] -// Activation is one of {relu, sigmoid, tanh} (loader allow-list). -void joint_step(const HostJoint & j, +// Activation is one of {relu, sigmoid, tanh} (loader allow-list). Returns +// false if the graph compute fails. +bool joint_step(const HostJoint & j, const JointGraph & g, const float * enc_proj, const float * pred_state, @@ -856,7 +861,10 @@ void joint_step(const HostJoint & j, // Full joint on the shared decoder pool, one graph, one dispatch. ggml_backend_tensor_set(g.pred_in, pred_state, 0, static_cast(j.pred_hidden) * sizeof(float)); ggml_backend_tensor_set(g.enc_in, enc_proj, 0, static_cast(j.joint_h) * sizeof(float)); - ggml_backend_graph_compute(g.backend, g.graph); + if (const ggml_status gs = ggml_backend_graph_compute(g.backend, g.graph); gs != GGML_STATUS_SUCCESS) { + log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet decoder: joint compute failed (%d)", static_cast(gs)); + return false; + } ggml_backend_tensor_get(g.logits, out_logits.data(), 0, static_cast(j.joint_n) * sizeof(float)); // log_softmax over the full joint output matches NeMo's CPU-inference @@ -882,6 +890,7 @@ void joint_step(const HostJoint & j, out_logits[i] -= log_sum; } } + return true; } // Compute the per-utterance encoder projection out[T, joint_h] = @@ -1094,7 +1103,10 @@ transcribe_status decode_tdt_greedy(const HostDecoderWeights & w, const int64_t t0 = ggml_time_us(); const float * decoder_out; if (predictor_dirty) { - decoder_out = predictor_step_ggml(w.predictor, pg, last_token, state, next_state, scratch_x); + decoder_out = predictor_step_ggml(w.predictor, pg, last_token, state, next_state, scratch_x); + if (decoder_out == nullptr) { + return TRANSCRIBE_ERR_BACKEND; + } predictor_dirty = false; } else { decoder_out = next_state.h.back().data(); @@ -1103,7 +1115,9 @@ transcribe_status decode_tdt_greedy(const HostDecoderWeights & w, // ----- Joint (using precomputed encoder projection) ----- const float * enc_proj = enc_proj_all.data() + static_cast(step) * static_cast(joint_h); - joint_step(w.joint, jg, enc_proj, decoder_out, logits); + if (!joint_step(w.joint, jg, enc_proj, decoder_out, logits)) { + return TRANSCRIBE_ERR_BACKEND; + } const int64_t t2 = ggml_time_us(); t_pred_us += t1 - t0; t_joint_us += t2 - t1; @@ -1297,7 +1311,10 @@ transcribe_status decode_rnnt_greedy(const HostDecoderWeights & w, const int64_t t0 = ggml_time_us(); const float * decoder_out; if (predictor_dirty) { - decoder_out = predictor_step_ggml(w.predictor, pg, last_token, state, next_state, scratch_x); + decoder_out = predictor_step_ggml(w.predictor, pg, last_token, state, next_state, scratch_x); + if (decoder_out == nullptr) { + return TRANSCRIBE_ERR_BACKEND; + } predictor_dirty = false; } else { decoder_out = next_state.h.back().data(); @@ -1305,7 +1322,9 @@ transcribe_status decode_rnnt_greedy(const HostDecoderWeights & w, const int64_t t1 = ggml_time_us(); const float * enc_proj = enc_proj_all.data() + static_cast(step) * static_cast(joint_h); - joint_step(w.joint, jg, enc_proj, decoder_out, logits); + if (!joint_step(w.joint, jg, enc_proj, decoder_out, logits)) { + return TRANSCRIBE_ERR_BACKEND; + } const int64_t t2 = ggml_time_us(); t_pred_us += t1 - t0; t_joint_us += t2 - t1; @@ -1475,14 +1494,19 @@ transcribe_status decode_rnnt_greedy_streaming(const HostDecoderWeights & w, const float * decoder_out; if (predictor_dirty) { - decoder_out = predictor_step_ggml(w.predictor, pg, last_token, state, next_state, scratch_x); + decoder_out = predictor_step_ggml(w.predictor, pg, last_token, state, next_state, scratch_x); + if (decoder_out == nullptr) { + return TRANSCRIBE_ERR_BACKEND; + } predictor_dirty = false; } else { decoder_out = next_state.h.back().data(); } const float * enc_proj = enc_proj_all.data() + static_cast(step) * static_cast(joint_h); - joint_step(w.joint, jg, enc_proj, decoder_out, logits); + if (!joint_step(w.joint, jg, enc_proj, decoder_out, logits)) { + return TRANSCRIBE_ERR_BACKEND; + } const float * token_logits = logits.data(); const int pred_token = argmax_range(token_logits, n_token_cls); diff --git a/src/arch/parakeet/model.cpp b/src/arch/parakeet/model.cpp index bc426aa7..52fef35e 100644 --- a/src/arch/parakeet/model.cpp +++ b/src/arch/parakeet/model.cpp @@ -215,7 +215,7 @@ transcribe_status fuse_batch_norm(ParakeetModel & m) { ggml_init_params params = { ctx_size, nullptr, /*no_alloc=*/true }; m.bn_fused_ctx = ggml_init(params); if (m.bn_fused_ctx == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } // Create all tensors first, then allocate a buffer. @@ -228,7 +228,7 @@ transcribe_status fuse_batch_norm(ParakeetModel & m) { // Allocate on the CPU backend (always last in the scheduler list). m.bn_fused_buffer = ggml_backend_alloc_ctx_tensors(m.bn_fused_ctx, m.plan.scheduler_list.back()); if (m.bn_fused_buffer == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } // Compute fused values from the raw BN tensors. @@ -351,7 +351,7 @@ transcribe_status init_streaming_caches(ParakeetSession * pc, ParakeetModel * pm log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet stream init_caches: backend buffer alloc failed"); ggml_free(pc->stream_caches.ctx); pc->stream_caches.ctx = nullptr; - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } pc->stream_caches.channel_len = 0; @@ -533,7 +533,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -1034,7 +1034,7 @@ transcribe_status run_one_shot_inner(ParakeetSession * pc, pc->compute_ctx = ggml_init(init_params); if (pc->compute_ctx == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet run: ggml_init for compute_ctx failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } } @@ -1084,13 +1084,13 @@ transcribe_status run_one_shot_inner(ParakeetSession * pc, /*graph_size=*/8192, /*parallel=*/false, /*op_offload=*/true); if (pc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet run: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(pc->sched); if (!ggml_backend_sched_alloc_graph(pc->sched, eb.graph)) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet run: ggml_backend_sched_alloc_graph failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } // Upload the mel; the row-major [num_mels, n_frames] buffer is @@ -1246,7 +1246,7 @@ transcribe_status run_one_shot_inner(ParakeetSession * pc, if (const ggml_status gs = ggml_backend_sched_graph_compute(pc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet run: ggml_backend_sched_graph_compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } pc->t_encode_us = ggml_time_us() - t_enc_start; pc->t_decode_us = 0; @@ -1416,7 +1416,7 @@ static transcribe_status run_batch_encode(ParakeetSession * init_params.no_alloc = true; pc->compute_ctx = ggml_init(init_params); if (pc->compute_ctx == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } } @@ -1440,12 +1440,12 @@ static transcribe_status run_batch_encode(ParakeetSession * static_cast(pm->plan.scheduler_list.size()), /*graph_size=*/8192, /*parallel=*/false, /*op_offload=*/true); if (pc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(pc->sched); if (!ggml_backend_sched_alloc_graph(pc->sched, eb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(eb.mel_in, pc->mel_buf.data(), 0, pc->mel_buf.size() * sizeof(float)); @@ -1580,7 +1580,7 @@ static transcribe_status run_batch_encode(ParakeetSession * const int64_t t_enc_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(pc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet run_batch: graph_compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } pc->t_encode_us = ggml_time_us() - t_enc_start; @@ -1880,7 +1880,7 @@ transcribe_status ensure_pos_proj_cache(ParakeetSession * pc, ParakeetModel * pm transcribe_status st = TRANSCRIBE_OK; ggml_backend_sched_reset(pc->sched); if (!ggml_backend_sched_alloc_graph(pc->sched, graph)) { - st = TRANSCRIBE_ERR_BACKEND; + st = TRANSCRIBE_ERR_OOM; } else { ggml_backend_tensor_set(pos_in, pc->pos_buf.data(), 0, static_cast(pos_len) * d_model * sizeof(float)); if (ggml_backend_sched_graph_compute(pc->sched, graph) != GGML_STATUS_SUCCESS) { @@ -2052,7 +2052,7 @@ transcribe_status emit_streaming_chunk(ParakeetSession * pc, ggml_backend_sched_reset(pc->sched); if (!ggml_backend_sched_alloc_graph(pc->sched, eb.graph)) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet stream: alloc_graph failed"); - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } // Upload mel chunk. Row-major [n_mels, n_mel_chunk_frames] is @@ -2126,7 +2126,7 @@ transcribe_status emit_streaming_chunk(ParakeetSession * pc, if (const ggml_status gs = ggml_backend_sched_graph_compute(pc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet stream: graph_compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } // Read encoder output back to host. @@ -2405,7 +2405,7 @@ transcribe_status emit_buffered_chunk(ParakeetSession * pc, pc->compute_ctx = ggml_init(init_params); if (pc->compute_ctx == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet buffered: ggml_init for compute_ctx failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } } @@ -2435,12 +2435,12 @@ transcribe_status emit_buffered_chunk(ParakeetSession * pc, static_cast(pm->plan.scheduler_list.size()), /*graph_size=*/8192, /*parallel=*/false, /*op_offload=*/true); if (pc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(pc->sched); if (!ggml_backend_sched_alloc_graph(pc->sched, eb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(eb.mel_in, pc->mel_buf.data(), 0, pc->mel_buf.size() * sizeof(float)); @@ -2503,7 +2503,7 @@ transcribe_status emit_buffered_chunk(ParakeetSession * pc, const int64_t t_enc_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(pc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet buffered: sched_graph_compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } pc->t_encode_us += ggml_time_us() - t_enc_start; diff --git a/src/arch/parakeet/multitalker.cpp b/src/arch/parakeet/multitalker.cpp index 9fd80874..5b9a29d1 100644 --- a/src/arch/parakeet/multitalker.cpp +++ b/src/arch/parakeet/multitalker.cpp @@ -388,7 +388,7 @@ transcribe_status run_multitalker(ParakeetSession * pc, /*graph_size=*/8192, /*parallel=*/false, /*op_offload=*/true); if (pc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "parakeet multitalker: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } diff --git a/src/arch/qwen3_asr/model.cpp b/src/arch/qwen3_asr/model.cpp index 250bef3e..53e02b61 100644 --- a/src/arch/qwen3_asr/model.cpp +++ b/src/arch/qwen3_asr/model.cpp @@ -249,7 +249,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "qwen3_asr: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -272,10 +272,12 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par for (auto & b : m->weights.dec_blocks) { entries.push_back({ b.ffn_gate_w, b.ffn_up_w, &b.ffn_gate_up_w }); } - if (!transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, - entries, m->packed_gate_up, "qwen3_asr")) { + if (const transcribe_status st = + transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, + entries, m->packed_gate_up, "qwen3_asr"); + st != TRANSCRIBE_OK) { m->packed_gate_up.free(); - return TRANSCRIBE_ERR_GGUF; + return st; } } @@ -621,7 +623,7 @@ transcribe_status run(transcribe_session * session, cc->compute_ctx = ggml_init(ip); if (cc->compute_ctx == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "qwen3_asr run: ggml_init for compute_ctx failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } } @@ -637,7 +639,7 @@ transcribe_status run(transcribe_session * session, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "qwen3_asr run: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -678,7 +680,7 @@ transcribe_status run(transcribe_session * session, t_enc_build_us = t_enc_start - t_enc_build_start; if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "qwen3_asr run: encoder graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_encode_us = ggml_time_us() - t_enc_start; @@ -831,7 +833,7 @@ transcribe_status run(transcribe_session * session, t_prefill_build_us = t_prefill_compute_start - t_prefill_build_start; if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, pb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "qwen3_asr run: prefill graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } t_prefill_compute_us = ggml_time_us() - t_prefill_compute_start; @@ -955,7 +957,7 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, sb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "qwen3_asr step: graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } const int64_t t_comp1 = ggml_time_us(); t_step_comp_us += t_comp1 - t_set1; @@ -1112,7 +1114,7 @@ transcribe_status reset_compute_ctx(QwenAsrSession * cc, int mb) { ip.mem_buffer = nullptr; ip.no_alloc = true; cc->compute_ctx = ggml_init(ip); - return cc->compute_ctx != nullptr ? TRANSCRIBE_OK : TRANSCRIBE_ERR_GGUF; + return cc->compute_ctx != nullptr ? TRANSCRIBE_OK : TRANSCRIBE_ERR_OOM; } // Batched encoder: mel (parallel) + one encoder graph over all B utterances @@ -1188,12 +1190,12 @@ transcribe_status encode_all_batched(QwenAsrSession * cc, cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, eb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } const int mel_per_chunk = cm->hparams.enc_n_window * 2; @@ -1265,7 +1267,7 @@ transcribe_status encode_all_batched(QwenAsrSession * cc, const int64_t t_enc0 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us += ggml_time_us() - t_enc0; @@ -1324,7 +1326,7 @@ transcribe_status prefill_all_batched(QwenAsrSession * ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, pb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } const int d_enc = cm->hparams.enc_output_dim; @@ -1401,7 +1403,7 @@ transcribe_status prefill_all_batched(QwenAsrSession * apply_sched_threads(cc); if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } std::vector amax(n, 0); @@ -1651,7 +1653,7 @@ transcribe_status run_batch(transcribe_session * session, } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, sb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } transcribe::causal_lm::StepBatchedIO io{}; diff --git a/src/arch/sensevoice/model.cpp b/src/arch/sensevoice/model.cpp index 95bf2403..4ceeb352 100644 --- a/src/arch/sensevoice/model.cpp +++ b/src/arch/sensevoice/model.cpp @@ -148,7 +148,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "sensevoice: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -462,11 +462,8 @@ transcribe_status run(transcribe_session * session, static_cast(cm->plan.scheduler_list.size()), /*graph_size=*/8192, /*parallel=*/false, /*op_offload=*/true); if (cc->sched == nullptr) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, - "sensevoice run: scheduler allocation failed — out of memory. " - "Split long audio into shorter segments (see " - "transcribe_capabilities.max_audio_ms)."); - return TRANSCRIBE_ERR_OOM; + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "sensevoice run: ggml_backend_sched_new failed"); + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -545,7 +542,7 @@ transcribe_status run(transcribe_session * session, const int64_t t_enc_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "sensevoice run: graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_encode_us = ggml_time_us() - t_enc_start; @@ -701,11 +698,8 @@ static transcribe_status run_batch_encode( static_cast(cm->plan.scheduler_list.size()), /*graph_size=*/8192, /*parallel=*/false, /*op_offload=*/true); if (cc->sched == nullptr) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, - "sensevoice run: scheduler allocation failed — out of memory. " - "Split long audio into shorter segments (see " - "transcribe_capabilities.max_audio_ms)."); - return TRANSCRIBE_ERR_OOM; + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "sensevoice run: ggml_backend_sched_new failed"); + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -776,7 +770,7 @@ static transcribe_status run_batch_encode( const int64_t t_enc_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "sensevoice run_batch: graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->t_encode_us = ggml_time_us() - t_enc_start; diff --git a/src/arch/sortformer/model.cpp b/src/arch/sortformer/model.cpp index 3544e9e8..d1c229a9 100644 --- a/src/arch/sortformer/model.cpp +++ b/src/arch/sortformer/model.cpp @@ -221,7 +221,7 @@ transcribe_status fuse_conformer_bn_core(std::vector & blocks ggml_init_params params = { ctx_size, nullptr, /*no_alloc=*/true }; *out_ctx = ggml_init(params); if (*out_ctx == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } for (size_t i = 0; i < n_blocks; ++i) { auto & b = blocks[i]; @@ -230,7 +230,7 @@ transcribe_status fuse_conformer_bn_core(std::vector & blocks } *out_buffer = ggml_backend_alloc_ctx_tensors(*out_ctx, alloc_backend); if (*out_buffer == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } std::vector bn_w(d), bn_b(d), rm(d), rv(d), fused_s(d), fused_b(d); for (size_t i = 0; i < n_blocks; ++i) { @@ -607,7 +607,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "sortformer: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -708,7 +708,7 @@ static transcribe_status ensure_sched(SortformerSession * pc, SortformerModel * static_cast(pm->plan.scheduler_list.size()), /*graph_size=*/8192, /*parallel=*/false, /*op_offload=*/true); if (pc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } return TRANSCRIBE_OK; @@ -730,7 +730,7 @@ static transcribe_status run_offline_forward(SortformerSession * pc, SortformerM ip.no_alloc = true; pc->compute_ctx = ggml_init(ip); if (pc->compute_ctx == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } } ggml_context * ctx = pc->compute_ctx; @@ -766,7 +766,7 @@ static transcribe_status run_offline_forward(SortformerSession * pc, SortformerM } ggml_backend_sched_reset(pc->sched); if (!ggml_backend_sched_alloc_graph(pc->sched, eb.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(eb.mel_in, pc->mel_buf.data(), 0, pc->mel_buf.size() * sizeof(float)); @@ -783,7 +783,7 @@ static transcribe_status run_offline_forward(SortformerSession * pc, SortformerM transcribe::configure_sched_n_threads(pc->sched, pc->n_threads); if (ggml_backend_sched_graph_compute(pc->sched, eb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "sortformer offline forward: graph_compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } transcribe::debug::dump_tensor("enc.fastconformer.out", eb.out, "encoder"); @@ -871,7 +871,7 @@ transcribe_status run_diar_streaming_core(DiarStreamScratch & sc, ip.no_alloc = true; sc.compute_ctx = ggml_init(ip); if (sc.compute_ctx == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } } ggml_context * ctx = sc.compute_ctx; @@ -896,14 +896,14 @@ transcribe_status run_diar_streaming_core(DiarStreamScratch & sc, ggml_backend_sched_reset(sched); if (!ggml_backend_sched_alloc_graph(sched, A.graph)) { transcribe::debug::pop_name_prefix(); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(A.mel_in, sc.chunk_mel_buf.data(), 0, sc.chunk_mel_buf.size() * sizeof(float)); transcribe::configure_sched_n_threads(sched, n_threads); if (ggml_backend_sched_graph_compute(sched, A.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "sortformer streaming: pre_encode compute failed"); transcribe::debug::pop_name_prefix(); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } sc.chunk_embs_host.resize(static_cast(T_diar) * ed); ggml_backend_tensor_get(A.out, sc.chunk_embs_host.data(), 0, sc.chunk_embs_host.size() * sizeof(float)); @@ -947,7 +947,7 @@ transcribe_status run_diar_streaming_core(DiarStreamScratch & sc, ggml_backend_sched_reset(sched); if (!ggml_backend_sched_alloc_graph(sched, B.graph)) { transcribe::debug::pop_name_prefix(); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(B.concat_in, sc.concat_host.data(), 0, sc.concat_host.size() * sizeof(float)); ggml_backend_tensor_set(B.pos_emb_in, sc.pos_buf.data(), 0, sc.pos_buf.size() * sizeof(float)); @@ -955,7 +955,7 @@ transcribe_status run_diar_streaming_core(DiarStreamScratch & sc, if (ggml_backend_sched_graph_compute(sched, B.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "sortformer streaming: infer compute failed"); transcribe::debug::pop_name_prefix(); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } sc.stream_preds_host.resize(static_cast(T_concat) * n_spk); ggml_backend_tensor_get(B.preds, sc.stream_preds_host.data(), 0, sc.stream_preds_host.size() * sizeof(float)); diff --git a/src/arch/voxtral/model.cpp b/src/arch/voxtral/model.cpp index aba6fa73..f817d8d5 100644 --- a/src/arch/voxtral/model.cpp +++ b/src/arch/voxtral/model.cpp @@ -166,7 +166,7 @@ transcribe_status prefill_chunked(VoxtralSession * cc, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, pb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral prefill: chunk %d/%d compute failed (%d)", c + 1, n_chunks, static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.n = max_n_kv; @@ -485,7 +485,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -505,10 +505,12 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par for (auto & b : m->weights.dec_blocks) { entries.push_back({ b.ffn_gate_w, b.ffn_up_w, &b.ffn_gate_up_w }); } - if (!transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, - entries, m->packed_gate_up, "voxtral")) { + if (const transcribe_status st = + transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, + entries, m->packed_gate_up, "voxtral"); + st != TRANSCRIBE_OK) { m->packed_gate_up.free(); - return TRANSCRIBE_ERR_GGUF; + return st; } } @@ -651,7 +653,7 @@ transcribe_status run(transcribe_session * session, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral run: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } @@ -702,7 +704,7 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral run: encoder compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } // Dump enc.* for the first chunk (matches the single-chunk reference). @@ -856,7 +858,7 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, pb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral run: prefill compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.n = T_prompt; cc->kv_cache.head = T_prompt; @@ -963,7 +965,7 @@ transcribe_status run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, sb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral run: step compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } // Mid-generation logits dump: the step at cur_past == T_prompt + 7 is // the reference's scores[8] (logits for the 9th generated token). @@ -1186,7 +1188,7 @@ transcribe_status run_batch(transcribe_session * session, cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } @@ -1248,7 +1250,7 @@ transcribe_status run_batch(transcribe_session * session, const int64_t t_enc0 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } enc_us = ggml_time_us() - t_enc0; @@ -1440,7 +1442,7 @@ transcribe_status run_batch(transcribe_session * session, const int64_t t_dec0 = ggml_time_us(); if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } dec_us += ggml_time_us() - t_dec0; diff --git a/src/arch/voxtral_realtime/model.cpp b/src/arch/voxtral_realtime/model.cpp index 6adfb758..a53d2ec4 100644 --- a/src/arch/voxtral_realtime/model.cpp +++ b/src/arch/voxtral_realtime/model.cpp @@ -266,7 +266,7 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral_realtime: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -287,10 +287,12 @@ transcribe_status load(Loader & loader, const transcribe_model_load_params * par for (auto & b : m->weights.dec_blocks) { entries.push_back({ b.ffn_gate_w, b.ffn_up_w, &b.ffn_gate_up_w }); } - if (!transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, - entries, m->packed_gate_up, "voxtral_realtime")) { + if (const transcribe_status st = + transcribe::causal_lm::pack_gate_up(m->plan.primary, m->hparams.dec_hidden, m->hparams.dec_intermediate, + entries, m->packed_gate_up, "voxtral_realtime"); + st != TRANSCRIBE_OK) { m->packed_gate_up.free(); - return TRANSCRIBE_ERR_GGUF; + return st; } } @@ -383,7 +385,7 @@ transcribe_status compute_ada_scales(Session * cc, Model * cm, int num_delay) { ggml_set_name(cc->ada_scale_all, "ada.scale_all"); cc->ada_buffer = ggml_backend_alloc_ctx_tensors(cc->ada_ctx, cm->plan.primary); if (cc->ada_buffer == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } // Build a one-shot compute graph: per layer ada_l = linear2(gelu(linear1(t))). @@ -418,7 +420,7 @@ transcribe_status compute_ada_scales(Session * cc, Model * cm, int num_delay) { static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { ggml_free(ctx); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -433,7 +435,7 @@ transcribe_status compute_ada_scales(Session * cc, Model * cm, int num_delay) { apply_threads(cc->sched, cc->n_threads); if (ggml_backend_sched_graph_compute(cc->sched, gf) != GGML_STATUS_SUCCESS) { ggml_free(ctx); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } std::vector ada(static_cast(hidden) * n_layer); @@ -535,7 +537,7 @@ transcribe_status forward_buffer(Session * cc, cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } @@ -607,7 +609,7 @@ transcribe_status forward_buffer(Session * cc, apply_threads(cc->sched, cc->n_threads); if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral_realtime run: encoder compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } if (dumps_on) { @@ -755,7 +757,7 @@ transcribe_status forward_buffer(Session * cc, apply_threads(cc->sched, cc->n_threads); if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral_realtime run: prefill compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.n = T_prompt; cc->kv_cache.head = T_prompt; @@ -830,7 +832,7 @@ transcribe_status forward_buffer(Session * cc, static_cast(max_n_kv) * sizeof(ggml_fp16_t)); if (ggml_backend_sched_graph_compute(cc->sched, sb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral_realtime run: step compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } int32_t tok = 0; ggml_backend_tensor_get(sb.out, &tok, 0, sizeof(int32_t)); @@ -941,7 +943,7 @@ transcribe_status forward_buffer(Session * cc, if (ggml_backend_sched_graph_compute(cc->sched, vb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral_realtime run: verify compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } ggml_backend_tensor_get(vb.out, predicted.data(), 0, static_cast(T_verify) * sizeof(int32_t)); @@ -1070,7 +1072,7 @@ transcribe_status forward_buffer(Session * cc, apply_threads(cc->sched, cc->n_threads); if (ggml_backend_sched_graph_compute(cc->sched, tf.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral_realtime run: teacher-forced dump compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } auto dump = [&](const char * n, ggml_tensor * t, const char * s) { @@ -1298,16 +1300,16 @@ std::string detok_generated(Model * cm, const std::vector & ids) { // Allocate + compute a graph on the session scheduler. Inputs are set by // `set_inputs` AFTER allocation (the alloc assigns tensor data pointers). -bool stream_run_graph(Session * cc, - Model * cm, - ggml_cgraph * gf, - const std::function & set_inputs, - int64_t * out_compute_us = nullptr) { +transcribe_status stream_run_graph(Session * cc, + Model * cm, + ggml_cgraph * gf, + const std::function & set_inputs, + int64_t * out_compute_us = nullptr) { if (cc->sched == nullptr) { cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return false; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -1315,16 +1317,21 @@ bool stream_run_graph(Session * cc, transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral_realtime: stream graph allocation failed — " "out of memory."); - return false; + return TRANSCRIBE_ERR_OOM; } set_inputs(); apply_threads(cc->sched, cc->n_threads); - const int64_t tc0 = ggml_time_us(); - const bool ok = ggml_backend_sched_graph_compute(cc->sched, gf) == GGML_STATUS_SUCCESS; + const int64_t tc0 = ggml_time_us(); + const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, gf); if (out_compute_us) { *out_compute_us = ggml_time_us() - tc0; // pure graph_compute } - return ok; + if (gs != GGML_STATUS_SUCCESS) { + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral_realtime: stream graph compute failed (%d)", + static_cast(gs)); + return TRANSCRIBE_ERR_BACKEND; + } + return TRANSCRIBE_OK; } // Encoder KV ring geometry. Keep the last `sliding_window`(750) frames: hold a @@ -1505,14 +1512,17 @@ transcribe_status stream_process(Session * cc, Model * cm, bool is_final, bool * cc->mel_buf[static_cast(k) * mel_n + (mrel + t)]; } } - if (!stream_run_graph(cc, cm, emb.graph, [&] { - ggml_backend_tensor_set(emb.mel_in, mel_fm.data(), 0, mel_fm.size() * sizeof(float)); - ggml_backend_tensor_set(emb.cache1_in, cc->stream_conv0_cache.data(), 0, - cc->stream_conv0_cache.size() * sizeof(float)); - ggml_backend_tensor_set(emb.cache2_in, cc->stream_conv1_cache.data(), 0, - cc->stream_conv1_cache.size() * sizeof(float)); - })) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = stream_run_graph( + cc, cm, emb.graph, + [&] { + ggml_backend_tensor_set(emb.mel_in, mel_fm.data(), 0, mel_fm.size() * sizeof(float)); + ggml_backend_tensor_set(emb.cache1_in, cc->stream_conv0_cache.data(), 0, + cc->stream_conv0_cache.size() * sizeof(float)); + ggml_backend_tensor_set(emb.cache2_in, cc->stream_conv1_cache.data(), 0, + cc->stream_conv1_cache.size() * sizeof(float)); + }); + st != TRANSCRIBE_OK) { + return st; } ggml_backend_tensor_get(emb.out, emb_host.data(), 0, emb_host.size() * sizeof(float)); // Carry conv caches for the next chunk: conv0 ← last 2 NEW mel frames @@ -1596,15 +1606,16 @@ transcribe_status stream_process(Session * cc, Model * cm, bool is_final, bool * } } int64_t enc_compute_us = 0; - if (!stream_run_graph( + if (const transcribe_status st = stream_run_graph( cc, cm, eb.graph, [&] { ggml_backend_tensor_set(eb.embed_in, chunk_in.data(), 0, chunk_in.size() * sizeof(float)); ggml_backend_tensor_set(eb.positions_in, pos.data(), 0, pos.size() * sizeof(int32_t)); ggml_backend_tensor_set(eb.mask_in, mask.data(), 0, mask.size() * sizeof(ggml_fp16_t)); }, - &enc_compute_us)) { - return TRANSCRIBE_ERR_GGUF; + &enc_compute_us); + st != TRANSCRIBE_OK) { + return st; } cc->stream_t_enc_compute_us += enc_compute_us; @@ -1683,14 +1694,17 @@ transcribe_status stream_process(Session * cc, Model * cm, bool is_final, bool * mask[static_cast(r) * T_prompt + c] = mz; } } - if (!stream_run_graph(cc, cm, pb.graph, [&] { - ggml_backend_tensor_set(pb.input_ids_in, prompt_ids.data(), 0, prompt_ids.size() * sizeof(int32_t)); - ggml_backend_tensor_set(pb.audio_in, cc->stream_audio_embeds.data(), 0, - static_cast(dec_h) * T_prompt * sizeof(float)); - ggml_backend_tensor_set(pb.positions_in, positions.data(), 0, positions.size() * sizeof(int32_t)); - ggml_backend_tensor_set(pb.mask_in, mask.data(), 0, mask.size() * sizeof(ggml_fp16_t)); - })) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = stream_run_graph( + cc, cm, pb.graph, + [&] { + ggml_backend_tensor_set(pb.input_ids_in, prompt_ids.data(), 0, prompt_ids.size() * sizeof(int32_t)); + ggml_backend_tensor_set(pb.audio_in, cc->stream_audio_embeds.data(), 0, + static_cast(dec_h) * T_prompt * sizeof(float)); + ggml_backend_tensor_set(pb.positions_in, positions.data(), 0, positions.size() * sizeof(int32_t)); + ggml_backend_tensor_set(pb.mask_in, mask.data(), 0, mask.size() * sizeof(ggml_fp16_t)); + }); + st != TRANSCRIBE_OK) { + return st; } cc->kv_cache.n = T_prompt; cc->kv_cache.head = T_prompt; @@ -1731,7 +1745,7 @@ transcribe_status stream_process(Session * cc, Model * cm, bool is_final, bool * cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } ggml_backend_sched_reset(cc->sched); @@ -1784,7 +1798,7 @@ transcribe_status stream_process(Session * cc, Model * cm, bool is_final, bool * ggml_backend_tensor_set(sb.mask_in, step_mask.data(), 0, static_cast(max_n_kv) * sizeof(ggml_fp16_t)); if (ggml_backend_sched_graph_compute(cc->sched, sb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } if (dumps_on && cur == T_prompt - 1 + 8) { // stream.logits_raw.gen8 cc->stream_gen8_logits.assign(vocab, 0.0f); @@ -2162,7 +2176,7 @@ transcribe_status run_batch_step_loop(Session * cc ggml_backend_tensor_set(sb.mask_in, mask_buf.data(), 0, mask_buf.size() * sizeof(ggml_fp16_t)); if (ggml_backend_sched_graph_compute(cc->sched, sb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral_realtime run_batch: step compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } ggml_backend_tensor_get(sb.out, out_buf.data(), 0, out_buf.size() * sizeof(int32_t)); @@ -2269,7 +2283,7 @@ transcribe_status run_batch(transcribe_session * session, cc->sched = ggml_backend_sched_new(cm->plan.scheduler_list.data(), nullptr, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } } @@ -2347,7 +2361,7 @@ transcribe_status run_batch(transcribe_session * session, apply_threads(cc->sched, cc->n_threads); if (ggml_backend_sched_graph_compute(cc->sched, eb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral_realtime run_batch: encoder compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } std::vector proj(static_cast(dec_h) * n_audio_max * n); @@ -2373,8 +2387,8 @@ transcribe_status run_batch(transcribe_session * session, const int64_t enc_us = ggml_time_us() - t_enc0; // ----- Pass 2: ada scales (shared; num_delay uniform across the batch) ----- - if (compute_ada_scales(cc, cm, num_delay) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = compute_ada_scales(cc, cm, num_delay); st != TRANSCRIBE_OK) { + return st; } // ----- Prompt (uniform) + short-clip gating ----- @@ -2535,7 +2549,7 @@ transcribe_status run_batch(transcribe_session * session, apply_threads(cc->sched, cc->n_threads); if (ggml_backend_sched_graph_compute(cc->sched, pb.graph) != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "voxtral_realtime run_batch: prefill compute failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } std::vector first(n, 0); ggml_backend_tensor_get(pb.out, first.data(), 0, first.size() * sizeof(int32_t)); diff --git a/src/arch/whisper/bin_load.cpp b/src/arch/whisper/bin_load.cpp index 1e1221e8..638fa627 100644 --- a/src/arch/whisper/bin_load.cpp +++ b/src/arch/whisper/bin_load.cpp @@ -591,7 +591,7 @@ transcribe_status load_from_bin(const char * path init_params.no_alloc = true; m->ctx_meta = ggml_init(init_params); if (m->ctx_meta == nullptr) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } std::vector stream_slots; @@ -620,7 +620,7 @@ transcribe_status load_from_bin(const char * path ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors(m->ctx_meta, m->plan.primary); if (buf == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "%s: ggml_backend_alloc_ctx_tensors failed", kTag); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = buf; ggml_backend_buffer_set_usage(buf, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); diff --git a/src/arch/whisper/model.cpp b/src/arch/whisper/model.cpp index ac2893ac..b7c82aea 100644 --- a/src/arch/whisper/model.cpp +++ b/src/arch/whisper/model.cpp @@ -481,7 +481,7 @@ transcribe_status whisper_load(Loader & loader, if (weights_buffer == nullptr) { gguf_free(gguf_data); log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper: ggml_backend_alloc_ctx_tensors failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } m->backend_buffer = weights_buffer; ggml_backend_buffer_set_usage(weights_buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -598,7 +598,7 @@ transcribe_status run_whisper_encoder_on_window(WhisperSession * cc, const int64_t t_enc_build_start = ggml_time_us(); if (!ensure_compute_ctx(cc, 8 * 1024 * 1024)) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run: ensure_compute_ctx (encoder) failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } EncoderBuild eb = build_encoder_graph(cc->compute_ctx, cm->weights, cm->hparams, n_mel_frames, @@ -617,7 +617,7 @@ transcribe_status run_whisper_encoder_on_window(WhisperSession * cc, if (cc->enc_out.tensor == nullptr || cc->enc_out.d_model != d_enc_g || cc->enc_out.T_enc != T_enc_g) { if (!enc_out_init(cc->enc_out, cm->plan.primary, d_enc_g, T_enc_g)) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run: enc_out_init failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } } ggml_tensor * enc_out_view = @@ -633,7 +633,7 @@ transcribe_status run_whisper_encoder_on_window(WhisperSession * cc, static_cast(cm->plan.scheduler_list.size()), 16384, false, true); if (cc->sched == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run: ggml_backend_sched_new failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } // Apply the caller's CPU thread count once at sched creation; it @@ -647,7 +647,7 @@ transcribe_status run_whisper_encoder_on_window(WhisperSession * cc, log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run: ggml_backend_sched_alloc_graph failed " "(encoder)"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } // Upload mel. @@ -659,7 +659,7 @@ transcribe_status run_whisper_encoder_on_window(WhisperSession * cc, const int64_t t_enc_compute_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, eb.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run: encoder graph compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->perf.enc_compute.add(ggml_time_us() - t_enc_compute_start); @@ -1666,7 +1666,7 @@ transcribe_status whisper_run(transcribe_session * session, ggml_backend_tensor_get(cc->enc_out.tensor, cc->enc_host.data(), 0, cc->enc_host.size() * sizeof(float)); if (!new_compute_ctx(16 * 1024 * 1024)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } DecoderBuild det_db = build_decoder_prefill_graph(cc->compute_ctx, cm->weights, cm->hparams, /*seq_len=*/1, T_enc_local, cc->decoder_use_flash); @@ -1675,7 +1675,7 @@ transcribe_status whisper_run(transcribe_session * session, } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, det_db.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } const int32_t sot = cm->hparams.decoder_start_token_id; ggml_backend_tensor_set(det_db.token_ids_in, &sot, 0, sizeof(int32_t)); @@ -1685,7 +1685,7 @@ transcribe_status whisper_run(transcribe_session * session, ggml_backend_tensor_set(det_db.causal_mask_in, &zero, 0, sizeof(float)); } if (ggml_backend_sched_graph_compute(cc->sched, det_db.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } const size_t row_bytes = static_cast(vocab_size) * sizeof(float); ggml_backend_tensor_get(det_db.dumps.logits_raw, last_logits.data(), 0, row_bytes); @@ -1889,7 +1889,7 @@ transcribe_status whisper_run(transcribe_session * session, if (!kv_cache_init(cc->kv_cache, cm->plan.primary, static_cast(n_ctx_decoder), T_enc_local, cm->hparams.dec_d_model, n_layers, kv_type_g)) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run: KV cache init failed"); - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } } @@ -1901,7 +1901,7 @@ transcribe_status whisper_run(transcribe_session * session, const int64_t t_cross_build_start = ggml_time_us(); if (!new_compute_ctx(8 * 1024 * 1024)) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run: ggml_init for cross_kv failed"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } DecoderBuild cross_db = build_cross_kv_graph(cc->compute_ctx, cm->weights, cm->hparams, cc->kv_cache, cc->enc_out.tensor, T_enc_local); @@ -1914,7 +1914,7 @@ transcribe_status whisper_run(transcribe_session * session, ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, cross_db.graph)) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run: alloc_graph failed (cross_kv)"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } // No tensor_set: cross-KV reads cc->enc_out.tensor via // a view inside build_cross_kv_graph, populated by the @@ -1925,7 +1925,7 @@ transcribe_status whisper_run(transcribe_session * session, if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, cross_db.graph); gs != GGML_STATUS_SUCCESS) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run: cross_kv compute failed (%d)", static_cast(gs)); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->perf.cross_compute.add(ggml_time_us() - t_cross_compute_start); cc->kv_cache.cross_populated = true; @@ -1959,7 +1959,7 @@ transcribe_status whisper_run(transcribe_session * session, { const int64_t t_prompt_build_start = ggml_time_us(); if (!new_compute_ctx(16 * 1024 * 1024)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } const int kv_pad = kv_pad_self_attn(cm->plan.primary_kind, cc->decoder_use_flash); DecoderBuild db = build_decoder_graph_kv(cc->compute_ctx, cm->weights, cm->hparams, cc->kv_cache, @@ -1975,7 +1975,7 @@ transcribe_status whisper_run(transcribe_session * session, ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, db.graph)) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run: alloc_graph failed (prompt)"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(db.token_ids_in, prompt_ids.data(), 0, prompt_ids.size() * sizeof(int32_t)); @@ -2020,7 +2020,7 @@ transcribe_status whisper_run(transcribe_session * session, const int64_t t_prompt_compute_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, db.graph); gs != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->perf.prompt_compute.add(ggml_time_us() - t_prompt_compute_start); cc->kv_cache.n = seq_len; @@ -2150,7 +2150,7 @@ transcribe_status whisper_run(transcribe_session * session, if (use_step_graph) { if (!new_compute_ctx(8 * 1024 * 1024)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } sb = build_step_graph(cc->compute_ctx, cm->weights, cm->hparams, cc->kv_cache, max_n_kv, T_enc_local, cc->decoder_use_flash); @@ -2161,7 +2161,7 @@ transcribe_status whisper_run(transcribe_session * session, ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, sb.graph)) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run: sched_alloc_graph failed (step)"); - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } // Self-attn mask: [0, seq_len) populated by prompt pass @@ -2222,7 +2222,7 @@ transcribe_status whisper_run(transcribe_session * session, const int64_t t_step_compute_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, sb.graph); gs != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->perf.step_compute.add(ggml_time_us() - t_step_compute_start); @@ -2236,7 +2236,7 @@ transcribe_status whisper_run(transcribe_session * session, } else { const int64_t t_step_build_start = ggml_time_us(); if (!new_compute_ctx(4 * 1024 * 1024)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } const int kv_pad = kv_pad_self_attn(cm->plan.primary_kind, cc->decoder_use_flash); DecoderBuild step_db = @@ -2252,7 +2252,7 @@ transcribe_status whisper_run(transcribe_session * session, const int64_t t_step_alloc_start = ggml_time_us(); ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, step_db.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } int32_t tok = next_id; @@ -2286,7 +2286,7 @@ transcribe_status whisper_run(transcribe_session * session, const int64_t t_step_compute_start = ggml_time_us(); if (const ggml_status gs = ggml_backend_sched_graph_compute(cc->sched, step_db.graph); gs != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->perf.step_compute.add(ggml_time_us() - t_step_compute_start); @@ -2783,7 +2783,7 @@ transcribe_status whisper_run_batch(transcribe_session * session, cc->enc_host.resize(enc_hosts[b].size()); std::memcpy(cc->enc_host.data(), enc_hosts[b].data(), enc_hosts[b].size() * sizeof(float)); if (!ensure_compute_ctx(cc, 16 * 1024 * 1024)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } DecoderBuild det = build_decoder_prefill_graph(cc->compute_ctx, cm->weights, hp, /*seq_len=*/1, T_enc_local, cc->decoder_use_flash); @@ -2792,7 +2792,7 @@ transcribe_status whisper_run_batch(transcribe_session * session, } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, det.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } const int32_t sot = hp.decoder_start_token_id; ggml_backend_tensor_set(det.token_ids_in, &sot, 0, sizeof(int32_t)); @@ -2802,7 +2802,7 @@ transcribe_status whisper_run_batch(transcribe_session * session, ggml_backend_tensor_set(det.causal_mask_in, &zero, 0, sizeof(float)); } if (ggml_backend_sched_graph_compute(cc->sched, det.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } std::vector ll(static_cast(vocab_size)); ggml_backend_tensor_get(det.dumps.logits_raw, ll.data(), 0, @@ -2896,7 +2896,7 @@ transcribe_status whisper_run_batch(transcribe_session * session, if (!kv_cache_init_batched(cc->kv_cache, cm->plan.primary, max_n_kv, T_enc_max, d_model, n_layer, B, kv_type_g)) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "whisper run_batch: kv_cache_init_batched failed"); - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } } else { ggml_backend_buffer_clear(cc->kv_cache.buffer, 0); @@ -2927,7 +2927,7 @@ transcribe_status whisper_run_batch(transcribe_session * session, // ---- Batched cross-attention K/V (tier-invariant; computed once). ---- { if (!new_compute_ctx(16 * 1024 * 1024)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } DecoderBuild cross = build_cross_kv_graph_batched(cc->compute_ctx, cm->weights, hp, cc->kv_cache, T_enc_max, B); if (cross.graph == nullptr) { @@ -2935,7 +2935,7 @@ transcribe_status whisper_run_batch(transcribe_session * session, } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, cross.graph)) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_OOM; } std::vector packed(static_cast(d_model) * T_enc_max * B, 0.0f); for (int b = 0; b < n; ++b) { @@ -2947,7 +2947,7 @@ transcribe_status whisper_run_batch(transcribe_session * session, } ggml_backend_tensor_set(cross.encoder_out_in, packed.data(), 0, packed.size() * sizeof(float)); if (ggml_backend_sched_graph_compute(cc->sched, cross.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } cc->kv_cache.cross_populated = true; } @@ -2972,24 +2972,24 @@ transcribe_status whisper_run_batch(transcribe_session * session, } StepBuildBatched sb{}; - auto rebuild_step = [&](int win) -> bool { + auto rebuild_step = [&](int win) -> transcribe_status { if (!new_compute_ctx(32 * 1024 * 1024)) { - return false; + return TRANSCRIBE_ERR_OOM; } sb = build_step_graph_batched(cc->compute_ctx, cm->weights, hp, cc->kv_cache, win, T_enc_max, B, cc->decoder_use_flash); if (sb.graph == nullptr || sb.logits_out == nullptr) { - return false; + return TRANSCRIBE_ERR_GGUF; } ggml_backend_sched_reset(cc->sched); if (!ggml_backend_sched_alloc_graph(cc->sched, sb.graph)) { - return false; + return TRANSCRIBE_ERR_OOM; } ggml_backend_tensor_set(sb.cross_mask_in, cmask.data(), 0, cmask.size() * sizeof(ggml_fp16_t)); - return true; + return TRANSCRIBE_OK; }; - if (!rebuild_step(kv_window)) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = rebuild_step(kv_window); st != TRANSCRIBE_OK) { + return st; } std::vector smask(static_cast(kv_window) * B, f16_ninf); @@ -3008,7 +3008,7 @@ transcribe_status whisper_run_batch(transcribe_session * session, ggml_backend_tensor_set(sb.kv_idx_in, kvidx_buf.data(), 0, B * sizeof(int64_t)); ggml_backend_tensor_set(sb.self_mask_in, smask.data(), 0, smask.size() * sizeof(ggml_fp16_t)); if (ggml_backend_sched_graph_compute(cc->sched, sb.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } return TRANSCRIBE_OK; }; @@ -3016,9 +3016,9 @@ transcribe_status whisper_run_batch(transcribe_session * session, ggml_backend_tensor_get(sb.logits_out, logits_host.data(), 0, static_cast(vocab_size) * B * sizeof(float)); }; - auto ensure_window = [&](int posv) -> bool { + auto ensure_window = [&](int posv) -> transcribe_status { if (posv + 1 <= kv_window) { - return true; + return TRANSCRIBE_OK; } int win = kv_window; while (win < posv + 1 && win < max_n_kv) { @@ -3028,7 +3028,7 @@ transcribe_status whisper_run_batch(transcribe_session * session, win = max_n_kv; } if (win == kv_window) { - return true; + return TRANSCRIBE_OK; } std::vector wider(static_cast(win) * B, f16_ninf); for (int b = 0; b < n; ++b) { @@ -3114,14 +3114,14 @@ transcribe_status whisper_run_batch(transcribe_session * session, if (cc->poll_abort()) { return TRANSCRIBE_ERR_ABORTED; } - if (!ensure_window(pos)) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = ensure_window(pos); st != TRANSCRIBE_OK) { + return st; } for (int b = 0; b < n; ++b) { tok_buf[b] = valid[b] ? prompts[b][pos] : eos_id; } - if (run_step(pos) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = run_step(pos); st != TRANSCRIBE_OK) { + return st; } if (ti == 0 && pos == sot_index && no_speech_token_id >= 0 && no_speech_token_id < static_cast(vocab_size)) { @@ -3184,14 +3184,14 @@ transcribe_status whisper_run_batch(transcribe_session * session, if (all_done || pos + 1 > max_n_kv) { break; } - if (!ensure_window(pos)) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = ensure_window(pos); st != TRANSCRIBE_OK) { + return st; } for (int b = 0; b < n; ++b) { tok_buf[b] = fin[b] ? eos_id : next_tok[b]; } - if (run_step(pos) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = run_step(pos); st != TRANSCRIBE_OK) { + return st; } read_logits(); for (int b = 0; b < n; ++b) { diff --git a/src/causal_lm/causal_lm.cpp b/src/causal_lm/causal_lm.cpp index 1da80bdf..e3736e57 100644 --- a/src/causal_lm/causal_lm.cpp +++ b/src/causal_lm/causal_lm.cpp @@ -808,16 +808,16 @@ void PackedGateUpHandles::free() { } } -bool pack_gate_up(ggml_backend_t backend, - int hidden, - int intermediate, - const std::vector & entries, - PackedGateUpHandles & out_handles, - const char * error_tag) { +transcribe_status pack_gate_up(ggml_backend_t backend, + int hidden, + int intermediate, + const std::vector & entries, + PackedGateUpHandles & out_handles, + const char * error_tag) { if (backend == nullptr || hidden <= 0 || intermediate <= 0 || entries.empty()) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "%s: pack_gate_up invalid args (hidden=%d intermediate=%d n=%zu)", error_tag, hidden, intermediate, entries.size()); - return false; + return TRANSCRIBE_ERR_GGUF; } const size_t ctx_size = entries.size() * ggml_tensor_overhead() + 1024; @@ -829,7 +829,7 @@ bool pack_gate_up(ggml_backend_t backend, out_handles.ctx = ggml_init(packed_params); if (out_handles.ctx == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "%s: pack_gate_up ggml_init failed", error_tag); - return false; + return TRANSCRIBE_ERR_OOM; } // One packed tensor per block, same dtype as gate_w (gate and up share @@ -838,17 +838,17 @@ bool pack_gate_up(ggml_backend_t backend, const auto & e = entries[i]; if (e.gate_w == nullptr || e.up_w == nullptr || e.gate_up_w_out == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "%s: pack_gate_up entry %zu has null member", error_tag, i); - return false; + return TRANSCRIBE_ERR_GGUF; } if (e.gate_w->type != e.up_w->type) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "%s: pack_gate_up entry %zu gate/up type mismatch (%d vs %d)", error_tag, i, static_cast(e.gate_w->type), static_cast(e.up_w->type)); - return false; + return TRANSCRIBE_ERR_GGUF; } ggml_tensor * t = ggml_new_tensor_2d(out_handles.ctx, e.gate_w->type, hidden, 2 * intermediate); if (t == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "%s: pack_gate_up new_tensor_2d failed at %zu", error_tag, i); - return false; + return TRANSCRIBE_ERR_OOM; } *e.gate_up_w_out = t; } @@ -856,7 +856,7 @@ bool pack_gate_up(ggml_backend_t backend, out_handles.buffer = ggml_backend_alloc_ctx_tensors(out_handles.ctx, backend); if (out_handles.buffer == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "%s: pack_gate_up backend buffer alloc failed", error_tag); - return false; + return TRANSCRIBE_ERR_OOM; } ggml_backend_buffer_set_usage(out_handles.buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS); @@ -870,7 +870,7 @@ bool pack_gate_up(ggml_backend_t backend, if (ggml_nbytes(gate_up) != gate_bytes + up_bytes) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "%s: pack_gate_up size mismatch (%zu vs %zu + %zu)", error_tag, ggml_nbytes(gate_up), gate_bytes, up_bytes); - return false; + return TRANSCRIBE_ERR_GGUF; } buf.resize(std::max(gate_bytes, up_bytes)); ggml_backend_tensor_get(e.gate_w, buf.data(), 0, gate_bytes); @@ -878,7 +878,7 @@ bool pack_gate_up(ggml_backend_t backend, ggml_backend_tensor_get(e.up_w, buf.data(), 0, up_bytes); ggml_backend_tensor_set(gate_up, buf.data(), gate_bytes, up_bytes); } - return true; + return TRANSCRIBE_OK; } transcribe_status run_batched_step_loop(transcribe_session * session, @@ -944,7 +944,7 @@ transcribe_status run_batched_step_loop(transcribe_session * sess ggml_backend_tensor_set(io.mask, mask_buf.data(), 0, mask_buf.size() * sizeof(ggml_fp16_t)); if (ggml_backend_sched_graph_compute(sched, io.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + return TRANSCRIBE_ERR_BACKEND; } ggml_backend_tensor_get(io.argmax, out_buf.data(), 0, out_buf.size() * sizeof(int32_t)); ++n_steps; diff --git a/src/causal_lm/causal_lm.h b/src/causal_lm/causal_lm.h index 46999d84..7d0b384f 100644 --- a/src/causal_lm/causal_lm.h +++ b/src/causal_lm/causal_lm.h @@ -293,14 +293,14 @@ struct PackedGateUpHandles { // Compatible with row-wise quants (Q4/Q5/Q6/Q8) because concat-along-dim-1 // is byte-concat for those types. Writes `*entries[i].gate_up_w_out` to the // new tensor and marks the buffer GGML_BACKEND_BUFFER_USAGE_WEIGHTS. -// Returns false on alloc / size-mismatch failure; `out_handles` is left in -// a state safe to free. -bool pack_gate_up(ggml_backend_t backend, - int hidden, - int intermediate, - const std::vector & entries, - PackedGateUpHandles & out_handles, - const char * error_tag = "causal_lm"); +// Returns TRANSCRIBE_ERR_OOM on allocation failure, TRANSCRIBE_ERR_GGUF on a +// gate/up shape or type mismatch; `out_handles` is left in a state safe to free. +transcribe_status pack_gate_up(ggml_backend_t backend, + int hidden, + int intermediate, + const std::vector & entries, + PackedGateUpHandles & out_handles, + const char * error_tag = "causal_lm"); // Batched greedy step loop (offline transcribe_run_batch decode). diff --git a/src/transcribe-batch-util.cpp b/src/transcribe-batch-util.cpp index 11372a9a..7ab16f14 100644 --- a/src/transcribe-batch-util.cpp +++ b/src/transcribe-batch-util.cpp @@ -217,8 +217,8 @@ transcribe_status run_batched_encdec_step_loop(transcribe_session * int kv_window = init_window; EncDecStepIO io{}; - if (!rebuild(kv_window, io)) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = rebuild(kv_window, io); st != TRANSCRIBE_OK) { + return st; } std::vector smask(static_cast(kv_window) * n, f16_ninf); @@ -243,8 +243,10 @@ transcribe_status run_batched_encdec_step_loop(transcribe_session * ggml_backend_tensor_set(io.pos_ids, pos_buf.data(), 0, n * sizeof(int32_t)); ggml_backend_tensor_set(io.kv_idx, kvidx_buf.data(), 0, n * sizeof(int64_t)); ggml_backend_tensor_set(io.self_mask, smask.data(), 0, smask.size() * sizeof(ggml_fp16_t)); - if (ggml_backend_sched_graph_compute(sched, io.graph) != GGML_STATUS_SUCCESS) { - return TRANSCRIBE_ERR_GGUF; + if (const ggml_status gs = ggml_backend_sched_graph_compute(sched, io.graph); gs != GGML_STATUS_SUCCESS) { + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "batched decode: step compute failed (%d)", + static_cast(gs)); + return TRANSCRIBE_ERR_BACKEND; } ggml_backend_tensor_get(io.argmax, argmax_buf.data(), 0, n * sizeof(int32_t)); ++n_steps; @@ -252,9 +254,9 @@ transcribe_status run_batched_encdec_step_loop(transcribe_session * }; // Grow the read window (rebuild graph + widen mask) so position `posv` fits. - auto ensure_window = [&](int posv) -> bool { + auto ensure_window = [&](int posv) -> transcribe_status { if (posv + 1 <= kv_window) { - return true; + return TRANSCRIBE_OK; } int win = kv_window; while (win < posv + 1 && win < max_n_kv) { @@ -264,7 +266,7 @@ transcribe_status run_batched_encdec_step_loop(transcribe_session * win = max_n_kv; } if (win == kv_window) { - return true; + return TRANSCRIBE_OK; } std::vector wider(static_cast(win) * n, f16_ninf); for (int b = 0; b < n; ++b) { @@ -282,17 +284,19 @@ transcribe_status run_batched_encdec_step_loop(transcribe_session * if (session->poll_abort()) { return TRANSCRIBE_ERR_ABORTED; } - if (!ensure_window(pos)) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, - "batched decode: step graph allocation failed — out of memory. " - "Lower transcribe_session_params.n_ctx or the batch size."); - return TRANSCRIBE_ERR_OOM; + if (const transcribe_status st = ensure_window(pos); st != TRANSCRIBE_OK) { + if (st == TRANSCRIBE_ERR_OOM) { + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, + "batched decode: step graph allocation failed — out of memory. " + "Lower transcribe_session_params.n_ctx or the batch size."); + } + return st; } for (int b = 0; b < n; ++b) { tok_buf[b] = prompt_ids[static_cast(pos)]; } - if (run_step(pos) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = run_step(pos); st != TRANSCRIBE_OK) { + return st; } } // argmax from the last prompt position = first generated token. @@ -323,17 +327,19 @@ transcribe_status run_batched_encdec_step_loop(transcribe_session * if (all_done || pos + 1 > max_n_kv) { break; } - if (!ensure_window(pos)) { - transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, - "batched decode: step graph allocation failed — out of memory. " - "Lower transcribe_session_params.n_ctx or the batch size."); - return TRANSCRIBE_ERR_OOM; + if (const transcribe_status st = ensure_window(pos); st != TRANSCRIBE_OK) { + if (st == TRANSCRIBE_ERR_OOM) { + transcribe::log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, + "batched decode: step graph allocation failed — out of memory. " + "Lower transcribe_session_params.n_ctx or the batch size."); + } + return st; } for (int b = 0; b < n; ++b) { tok_buf[b] = finished[b] ? eos_id : next_tok[b]; } - if (run_step(pos) != TRANSCRIBE_OK) { - return TRANSCRIBE_ERR_GGUF; + if (const transcribe_status st = run_step(pos); st != TRANSCRIBE_OK) { + return st; } for (int b = 0; b < n; ++b) { if (finished[b]) { diff --git a/src/transcribe-batch-util.h b/src/transcribe-batch-util.h index 9be59eed..961a64ae 100644 --- a/src/transcribe-batch-util.h +++ b/src/transcribe-batch-util.h @@ -128,9 +128,10 @@ struct EncDecStepIO { // Build (or rebuild) the step graph for self-attention window `win`: allocate a // fresh compute graph, build it, reset+alloc the scheduler, upload the static -// cross-attention mask, and fill `io`. Returns false on any failure. Called once -// at the initial window and again whenever the window must grow. -using EncDecRebuildFn = std::function; +// cross-attention mask, and fill `io`. Returns the failure status (OOM for an +// allocation failure) or TRANSCRIBE_OK. Called once at the initial window and +// again whenever the window must grow. +using EncDecRebuildFn = std::function; // Run the shared greedy enc-dec step loop. Feeds `prompt_ids[0..prompt_len)` as // uniform lockstep tokens, then generates until each row emits eos_id, the batch diff --git a/src/transcribe-load-common.cpp b/src/transcribe-load-common.cpp index 4b9bc99a..0f17b856 100644 --- a/src/transcribe-load-common.cpp +++ b/src/transcribe-load-common.cpp @@ -520,7 +520,7 @@ transcribe_status promote_conv_pw_f16_to_f32_on_cpu(const BackendPlan & ggml_init_params params = { ctx_size, nullptr, true }; ggml_context * ctx = ggml_init(params); if (ctx == nullptr) { - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } // Allocate F32 replacements in the new ctx, matching each source's @@ -531,7 +531,7 @@ transcribe_status promote_conv_pw_f16_to_f32_on_cpu(const BackendPlan & ggml_tensor * r = ggml_new_tensor(ctx, GGML_TYPE_F32, ggml_n_dims(s.src), s.src->ne); if (r == nullptr) { ggml_free(ctx); - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } ggml_set_name(r, s.src->name); replacements.push_back(r); @@ -541,7 +541,7 @@ transcribe_status promote_conv_pw_f16_to_f32_on_cpu(const BackendPlan & if (buffer == nullptr) { log_msg(TRANSCRIBE_LOG_LEVEL_ERROR, "%s: conv_pw f32 promotion buffer alloc failed", error_tag); ggml_free(ctx); - return TRANSCRIBE_ERR_BACKEND; + return TRANSCRIBE_ERR_OOM; } ggml_backend_buffer_set_usage(buffer, GGML_BACKEND_BUFFER_USAGE_WEIGHTS);