cwenzi/neuroflow-cpp
1
1#include "neuroflow/adamw.hpp"
2
3#include <cmath>
4#include <iostream>
5
6namespace neuroflow {
7
8AdamW::AdamW(float lr, float beta1, float beta2, float eps, float weight_decay)
9 : lr_(lr), beta1_(beta1), beta2_(beta2), eps_(eps),
10 weight_decay_(weight_decay), step_(0) {}
11
12void AdamW::add_param_group(const ParamGroup& group) {
13 param_groups_.push_back(group);
14 std::vector<ParamState> group_states;
15 for (size_t i = 0; i < group.params.size(); ++i) {
16 ParamState ps;
17 ps.m = Tensor(group.params[i]->shape_, QuantType::FP32);
18 ps.v = Tensor(group.params[i]->shape_, QuantType::FP32);
19 memset(ps.m.as_fp32(), 0, ps.m.data_size_);
20 memset(ps.v.as_fp32(), 0, ps.v.data_size_);
21 group_states.push_back(std::move(ps));
22 }
23 states_.push_back(std::move(group_states));
24}
25
26void AdamW::step() {
27 step_++;
28 double bias_corr1 = 1.0 - std::pow(static_cast<double>(beta1_), static_cast<double>(step_));
29 double bias_corr2 = 1.0 - std::pow(static_cast<double>(beta2_), static_cast<double>(step_));
30
31 for (size_t g = 0; g < param_groups_.size(); ++g) {
32 auto& group = param_groups_[g];
33 auto& gstates = states_[g];
34 float group_lr = group.lr > 0 ? group.lr : lr_;
35 float group_wd = group.weight_decay;
36
37 for (size_t i = 0; i < group.params.size(); ++i) {
38 Tensor* param = group.params[i];
39 Tensor* grad = group.grads[i];
40 if (!param || !grad || param->numel() == 0 || grad->numel() == 0) continue;
41
42 float* p = param->as_fp32();
43 const float* g = grad->as_fp32();
44 float* m = gstates[i].m.as_fp32();
45 float* v = gstates[i].v.as_fp32();
46 size_t n = param->numel();
47
48 for (size_t j = 0; j < n; ++j) {
49 if (!std::isfinite(g[j])) continue;
50
51 p[j] -= group_lr * group_wd * p[j];
52
53 m[j] = beta1_ * m[j] + (1.0f - beta1_) * g[j];
54 v[j] = beta2_ * v[j] + (1.0f - beta2_) * g[j] * g[j];
55
56 float m_hat = static_cast<float>(static_cast<double>(m[j]) / bias_corr1);
57 float v_hat = static_cast<float>(static_cast<double>(v[j]) / bias_corr2);
58
59 p[j] -= group_lr * m_hat / (std::sqrt(v_hat) + eps_);
60 }
61 }
62 }
63}
64
65void AdamW::set_lr(float lr) { lr_ = lr; }
66
67} // namespace neuroflow