Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions ggml/src/ggml-metal/ggml-metal-context.m
Original file line number Diff line number Diff line change
Expand Up @@ -79,6 +79,10 @@
// error state - set when a command buffer fails during synchronize
// once set, graph_compute will return GGML_STATUS_FAILED until the backend is recreated
bool has_error;

// lookups that found no buffer while THIS context encoded its current graph (counted by the
// encode blocks through ggml_metal_nil_sink_set; read once every block has finished)
uint64_t nil_lookups;
};

ggml_metal_t ggml_metal_init(ggml_metal_device_t dev) {
Expand Down Expand Up @@ -441,6 +445,10 @@ enum ggml_status ggml_metal_graph_compute(ggml_metal_t ctx, struct ggml_cgraph *
return GGML_STATUS_FAILED;
}

// a lookup that finds no buffer during this graph's encode means some op ran on the wrong
// memory: the graph is reported failed, never as a success computed on stale bytes
__atomic_store_n(&ctx->nil_lookups, 0, __ATOMIC_RELAXED);

// number of nodes encoded by the main thread (empirically determined)
const int n_main = MAX(64, 0.1*gf->n_nodes);

Expand Down Expand Up @@ -611,6 +619,12 @@ enum ggml_status ggml_metal_graph_compute(ggml_metal_t ctx, struct ggml_cgraph *
}
}

if (__atomic_load_n(&ctx->nil_lookups, __ATOMIC_RELAXED) != 0) {
GGML_LOG_ERROR("%s: %llu tensor lookup(s) found no buffer while encoding this graph: its result is not trusted\n",
__func__, (unsigned long long) __atomic_load_n(&ctx->nil_lookups, __ATOMIC_RELAXED));
return GGML_STATUS_FAILED;
}

return GGML_STATUS_SUCCESS;
}

Expand Down Expand Up @@ -704,6 +718,7 @@ void ggml_metal_set_n_cb(ggml_metal_t ctx, int n_cb) {
ctx->debug_graph,
ctx->debug_fusion);

ggml_metal_nil_sink_set(&ctx->nil_lookups); // this block's lookups count against this graph
for (int idx = 0; idx < ggml_metal_op_n_nodes(ctx_op); ++idx) {
const int res = ggml_metal_op_encode(ctx_op, idx);
if (res == 0) {
Expand All @@ -712,6 +727,7 @@ void ggml_metal_set_n_cb(ggml_metal_t ctx, int n_cb) {

idx += res - 1;
}
ggml_metal_nil_sink_set(NULL);

ggml_metal_op_free(ctx_op);

Expand Down
3 changes: 3 additions & 0 deletions ggml/src/ggml-metal/ggml-metal-device.h
Original file line number Diff line number Diff line change
Expand Up @@ -346,6 +346,9 @@ void ggml_metal_buffer_clear (ggml_metal_buffer_t buf, uint8_t value);
// Metal buffer based on the host memory pointer
//
struct ggml_metal_buffer_id ggml_metal_buffer_get_id(ggml_metal_buffer_t buf, const struct ggml_tensor * t);
// where this thread's lookups that find no buffer are counted (the graph compute it encodes
// for; NULL: not counted). See ggml_metal_buffer_get_id.
void ggml_metal_nil_sink_set(uint64_t * sink);

// [MOE-GATHER #23] GPU virtual address of a host pointer inside this Metal buffer (MTLBuffer.gpuAddress
// + intra-buffer delta), or 0 when the pointer is not covered / the OS lacks gpuAddress. Cross-buffer
Expand Down
14 changes: 14 additions & 0 deletions ggml/src/ggml-metal/ggml-metal-device.m
Original file line number Diff line number Diff line change
Expand Up @@ -2378,6 +2378,17 @@ void ggml_metal_buffer_clear(ggml_metal_buffer_t buf, uint8_t value) {
}
}

// Lookups that found no buffer holding a tensor's bytes are counted into the graph compute
// that is encoding on this thread (ggml_metal_nil_sink_set): such a lookup hands the kernel a
// nil buffer, the op reads or writes nothing it was meant to, and the graph goes on with stale
// memory (Metal OUT_PROD read stale scratch in every LoRA backward that way, one log line each).
// Per compute, never process-wide: a training graph's lookup must not fail a decode beside it.
static _Thread_local uint64_t * g_ggml_metal_nil_sink = NULL;

void ggml_metal_nil_sink_set(uint64_t * sink) {
g_ggml_metal_nil_sink = sink;
}

struct ggml_metal_buffer_id ggml_metal_buffer_get_id(ggml_metal_buffer_t buf, const struct ggml_tensor * t) {
struct ggml_metal_buffer_id res = { nil, 0 };

Expand All @@ -2399,6 +2410,9 @@ struct ggml_metal_buffer_id ggml_metal_buffer_get_id(ggml_metal_buffer_t buf, co
}

GGML_LOG_ERROR("%s: error: tensor '%s' buffer is nil\n", __func__, t->name);
if (g_ggml_metal_nil_sink != NULL) {
__atomic_fetch_add(g_ggml_metal_nil_sink, 1, __ATOMIC_RELAXED);
}

return res;
}
Expand Down
9 changes: 8 additions & 1 deletion ggml/src/ggml-opt.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1090,7 +1090,14 @@ void ggml_opt_eval(ggml_opt_context_t opt_ctx, ggml_opt_result_t result) {
}
}

ggml_backend_sched_graph_compute(opt_ctx->backend_sched, opt_ctx->allocated_graph_copy);
const enum ggml_status status = ggml_backend_sched_graph_compute(opt_ctx->backend_sched, opt_ctx->allocated_graph_copy);
if (status != GGML_STATUS_SUCCESS) {
// the step's gradients (and the optimizer update inside this graph) were computed on
// memory the backend says it did not have: the run must stop, never continue on them
opt_ctx->refusal = std::string("the backend failed to compute the training graph (") + ggml_status_to_string(status)
+ "): this step's gradients are not trusted, so the run stopped and nothing was written";
GGML_LOG_ERROR("%s: %s\n", __func__, opt_ctx->refusal.c_str());
}
opt_ctx->iter += opt_ctx->allocated_graph == opt_ctx->gb_opt;
opt_ctx->opt_i = (opt_ctx->opt_i + 1) % opt_ctx->opt_period;

Expand Down
9 changes: 9 additions & 0 deletions src/llama-context.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -3537,6 +3537,15 @@ void llama_context::opt_epoch_iter(
GGML_ASSERT(row == n_outputs);
}
ggml_opt_eval(opt_ctx, result);
if (const char * why = ggml_opt_refusal(opt_ctx); why[0] != '\0') {
// the backend failed the graph (ggml_opt_eval): the run fails, as a refused graph does
opt_failure = why;
LLAMA_LOG_ERROR("%s: %s: stopping the epoch\n", __func__, why);
ggml_free(ctx_compute_opt);
opt_alloc_failed.store(true);
opt_stop_requested.store(true);
return;
}
if (callback) {
callback(train, opt_ctx, dataset, result, idata_in_loop + (pos_ctx + pos_batch)/n_ubatch + 1, ndata_in_loop, t_loop_start);
}
Expand Down
Loading