diff --git a/ggml/src/ggml-opt.cpp b/ggml/src/ggml-opt.cpp index 5220ec5f65b4..b709375b52d6 100644 --- a/ggml/src/ggml-opt.cpp +++ b/ggml/src/ggml-opt.cpp @@ -906,6 +906,21 @@ bool ggml_opt_alloc(ggml_opt_context_t opt_ctx, bool backward) { if (!opt_ctx->static_graphs) { ggml_opt_build(opt_ctx); + + // Graphs built per step keep their gradient accumulators in ctx_static across graphs, + // and the reset above found no graph to reset (gb_grad is rebuilt every step). Without + // this, a period's step applied the SUM of every gradient since the run began, since + // each backward adds into the accumulator in place (test-opt-dynamic-accum). + if (backward && opt_ctx->opt_i == 0) { + // grad_accs is indexed by the FIRST graph's nodes: a later graph may hold fewer (Fable) + const size_t n = std::min(opt_ctx->grad_accs.size(), (size_t) opt_ctx->gf->n_nodes); + for (size_t i = 0; i < n; ++i) { + ggml_tensor * acc = opt_ctx->grad_accs[i]; + if (acc && (opt_ctx->gf->nodes[i]->flags & GGML_TENSOR_FLAG_PARAM)) { + ggml_set_zero(acc); + } + } + } } struct ggml_cgraph * graph = nullptr; diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index c33052830122..e04d5bf4cd17 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -150,6 +150,9 @@ endif () llama_build(test-recurrent-state-rollback.cpp) +# ggml-opt with per-step graphs (the llama training path) starts each optimizer period from zero accumulators +llama_build_and_test(test-opt-dynamic-accum.cpp) + if (NOT WIN32 OR NOT BUILD_SHARED_LIBS) # these tests are disabled on Windows because they use internal functions not exported with LLAMA_API (when building with shared libraries) llama_build_and_test(test-unicode.cpp) diff --git a/tests/test-opt-dynamic-accum.cpp b/tests/test-opt-dynamic-accum.cpp new file mode 100644 index 000000000000..03f857d27596 --- /dev/null +++ b/tests/test-opt-dynamic-accum.cpp @@ -0,0 +1,95 @@ +// what this catches: ggml-opt with graphs built per step (ggml_opt_prepare_alloc, the path +// llama_context's training takes) must start every optimizer period from ZERO gradient +// accumulators. Each step here is SGD on loss = sum(w * x), so its gradient is x and each +// step moves w by exactly -lr * x. If the accumulators carried the previous period's gradient, +// the second step would move w by -2 lr x, the third by -3 lr x. + +#include "ggml.h" +#include "ggml-alloc.h" +#include "ggml-backend.h" +#include "ggml-opt.h" + +#include +#include +#include + +#define CHECK(cond) do { if (!(cond)) { fprintf(stderr, "FAILED %s:%d: %s\n", __FILE__, __LINE__, #cond); exit(1); } } while (0) + +static const float LR = 0.125f; +static const float X = 2.0f; + +static ggml_opt_optimizer_params sgd_pars(void *) { + ggml_opt_optimizer_params p = ggml_opt_get_default_optimizer_params(nullptr); + p.sgd.alpha = LR; + p.sgd.wd = 0.0f; + return p; +} + +static void run(int32_t opt_period) { + // by type, not ggml_backend_cpu_init: in a dynamically loaded backend build (CI) the CPU + // backend is its own library + ggml_backend_t cpu = ggml_backend_init_by_type(GGML_BACKEND_DEVICE_TYPE_CPU, nullptr); + GGML_ASSERT(cpu != nullptr); + ggml_backend_t backends[] = { cpu }; + ggml_backend_sched_t sched = ggml_backend_sched_new(backends, nullptr, 1, GGML_DEFAULT_GRAPH_SIZE, false, true); + + ggml_init_params sp = { 8 * ggml_tensor_overhead(), nullptr, true }; + ggml_context * ctx_static = ggml_init(sp); + ggml_tensor * w = ggml_new_tensor_1d(ctx_static, GGML_TYPE_F32, 1); + ggml_set_param(w); + ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors(ctx_static, cpu); + float w_now = 1.0f; + ggml_backend_tensor_set(w, &w_now, 0, sizeof(float)); + + ggml_opt_params params = ggml_opt_default_params(sched, GGML_OPT_LOSS_TYPE_SUM); + params.optimizer = GGML_OPT_OPTIMIZER_TYPE_SGD; + params.get_opt_pars = sgd_pars; + params.opt_period = opt_period; + ggml_opt_context_t opt_ctx = ggml_opt_init(params); // no ctx_compute: graphs are built per step + ggml_opt_result_t result = ggml_opt_result_init(); + + const int n_evals = 3 * opt_period; + for (int i = 0; i < n_evals; ++i) { + ggml_init_params cp = { GGML_DEFAULT_GRAPH_SIZE * ggml_tensor_overhead() + 4 * ggml_graph_overhead_custom(GGML_DEFAULT_GRAPH_SIZE, true), nullptr, true }; + ggml_context * ctx_compute = ggml_init(cp); + ggml_tensor * x = ggml_new_tensor_1d(ctx_compute, GGML_TYPE_F32, 1); + ggml_tensor * out = ggml_mul(ctx_compute, w, x); + ggml_cgraph * gf = ggml_new_graph_custom(ctx_compute, GGML_DEFAULT_GRAPH_SIZE, true); + ggml_build_forward_expand(gf, out); + ggml_opt_prepare_alloc(opt_ctx, ctx_compute, gf, x, out); + CHECK(ggml_opt_alloc(opt_ctx, true)); + ggml_backend_tensor_set(x, &X, 0, sizeof(float)); + ggml_opt_eval(opt_ctx, result); + ggml_free(ctx_compute); + + float w_after; + ggml_backend_tensor_get(w, &w_after, 0, sizeof(float)); + if ((i + 1) % opt_period == 0) { + // one period = opt_period evals of gradient X each, scaled by nothing (LOSS_TYPE_SUM) + const float expected = w_now - LR * X * opt_period; + if (std::fabs(w_after - expected) > 1e-6f) { + fprintf(stderr, "opt_period %d, step %d: w moved to %f, expected %f (from %f)\n", + opt_period, (i + 1) / opt_period, w_after, expected, w_now); + exit(1); + } + w_now = w_after; + } else { + CHECK(std::fabs(w_after - w_now) < 1e-7f && "no update inside a period"); + } + } + + ggml_opt_result_free(result); + ggml_opt_free(opt_ctx); + ggml_backend_buffer_free(buf); + ggml_free(ctx_static); + ggml_backend_sched_free(sched); + ggml_backend_free(cpu); +} + +int main() { + ggml_backend_load_all(); + run(1); + run(2); + printf("test-opt-dynamic-accum: OK\n"); + return 0; +}