mirror of
https://github.com/leejet/stable-diffusion.cpp.git
synced 2026-10-02 10:24:37 +08:00
Compare commits
4
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
ca33304318 | ||
|
|
69efe3ce2b | ||
|
|
2eac844bbd | ||
|
|
968226abb2 |
@@ -82,7 +82,7 @@ git submodule update
|
||||
```shell
|
||||
curl -L -O https://huggingface.co/CompVis/stable-diffusion-v-1-4-original/resolve/main/sd-v1-4.ckpt
|
||||
# curl -L -O https://huggingface.co/runwayml/stable-diffusion-v1-5/resolve/main/v1-5-pruned-emaonly.safetensors
|
||||
# curl -L -O https://huggingface.co/stabilityai/stable-diffusion-2-1/blob/main/v2-1_768-nonema-pruned.safetensors
|
||||
# curl -L -O https://huggingface.co/stabilityai/stable-diffusion-2-1/resolve/main/v2-1_768-nonema-pruned.safetensors
|
||||
```
|
||||
|
||||
### Build
|
||||
|
||||
@@ -1102,6 +1102,7 @@ bool ModelLoader::parse_data_pkl(uint8_t* buffer,
|
||||
reader.tensor_storage.file_index = file_index;
|
||||
reader.tensor_storage.name = prefix + reader.tensor_storage.name;
|
||||
tensor_storages.push_back(reader.tensor_storage);
|
||||
// LOG_DEBUG("%s", reader.tensor_storage.name.c_str());
|
||||
// reset
|
||||
reader = PickleTensorReader();
|
||||
}
|
||||
@@ -1139,7 +1140,7 @@ bool ModelLoader::init_from_ckpt_file(const std::string& file_path, const std::s
|
||||
size_t pkl_size;
|
||||
zip_entry_read(zip, &pkl_data, &pkl_size);
|
||||
|
||||
LOG_DEBUG("%lld", pkl_size);
|
||||
// LOG_DEBUG("%lld", pkl_size);
|
||||
|
||||
parse_data_pkl((uint8_t*)pkl_data, pkl_size, zip, dir, file_index, prefix);
|
||||
|
||||
|
||||
@@ -7,8 +7,8 @@
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include "ggml/ggml.h"
|
||||
#include "ggml/ggml-backend.h"
|
||||
#include "ggml/ggml.h"
|
||||
#include "json.hpp"
|
||||
#include "zip.h"
|
||||
|
||||
|
||||
+205
-278
@@ -131,7 +131,7 @@ void print_ggml_tensor(struct ggml_tensor* tensor, bool shape_only = false) {
|
||||
if (shape_only) {
|
||||
return;
|
||||
}
|
||||
int range = 1000;
|
||||
int range = 3;
|
||||
for (int i = 0; i < tensor->ne[3]; i++) {
|
||||
if (i >= range && i + range < tensor->ne[3]) {
|
||||
continue;
|
||||
@@ -335,7 +335,7 @@ void sd_image_to_tensor(const uint8_t* image_data,
|
||||
}
|
||||
}
|
||||
|
||||
float sd_mean(struct ggml_tensor* src) {
|
||||
float ggml_tensor_mean(struct ggml_tensor* src) {
|
||||
float mean = 0.0f;
|
||||
int64_t nelements = ggml_nelements(src);
|
||||
float* data = (float*)src->data;
|
||||
@@ -345,7 +345,18 @@ float sd_mean(struct ggml_tensor* src) {
|
||||
return mean;
|
||||
}
|
||||
|
||||
void sd_scale(struct ggml_tensor* src, float scale) {
|
||||
// a = a+b
|
||||
void ggml_tensor_add(struct ggml_tensor* a, struct ggml_tensor* b) {
|
||||
GGML_ASSERT(ggml_nelements(a) == ggml_nelements(b));
|
||||
int64_t nelements = ggml_nelements(a);
|
||||
float* vec_a = (float*)a->data;
|
||||
float* vec_b = (float*)b->data;
|
||||
for (int i = 0; i < nelements; i++) {
|
||||
vec_a[i] = vec_a[i] + vec_b[i];
|
||||
}
|
||||
}
|
||||
|
||||
void ggml_tensor_scale(struct ggml_tensor* src, float scale) {
|
||||
int64_t nelements = ggml_nelements(src);
|
||||
float* data = (float*)src->data;
|
||||
for (int i = 0; i < nelements; i++) {
|
||||
@@ -353,7 +364,7 @@ void sd_scale(struct ggml_tensor* src, float scale) {
|
||||
}
|
||||
}
|
||||
|
||||
void sd_clamp(struct ggml_tensor* src, float min, float max) {
|
||||
void ggml_tensor_clamp(struct ggml_tensor* src, float min, float max) {
|
||||
int64_t nelements = ggml_nelements(src);
|
||||
float* data = (float*)src->data;
|
||||
for (int i = 0; i < nelements; i++) {
|
||||
@@ -363,7 +374,7 @@ void sd_clamp(struct ggml_tensor* src, float min, float max) {
|
||||
}
|
||||
|
||||
// convert values from [0, 1] to [-1, 1]
|
||||
void sd_convert_input(struct ggml_tensor* src) {
|
||||
void ggml_tensor_scale_input(struct ggml_tensor* src) {
|
||||
int64_t nelements = ggml_nelements(src);
|
||||
float* data = (float*)src->data;
|
||||
for (int i = 0; i < nelements; i++) {
|
||||
@@ -373,7 +384,7 @@ void sd_convert_input(struct ggml_tensor* src) {
|
||||
}
|
||||
|
||||
// convert values from [-1, 1] to [0, 1]
|
||||
void sd_convert_output(struct ggml_tensor* src) {
|
||||
void ggml_tensor_scale_output(struct ggml_tensor* src) {
|
||||
int64_t nelements = ggml_nelements(src);
|
||||
float* data = (float*)src->data;
|
||||
for (int i = 0; i < nelements; i++) {
|
||||
@@ -387,6 +398,64 @@ struct ggml_tensor* ggml_group_norm_32(struct ggml_context* ctx,
|
||||
return ggml_group_norm(ctx, a, 32);
|
||||
}
|
||||
|
||||
struct ggml_tensor* ggml_nn_linear(struct ggml_context* ctx,
|
||||
struct ggml_tensor* x,
|
||||
struct ggml_tensor* w,
|
||||
struct ggml_tensor* b) {
|
||||
x = ggml_mul_mat(ctx, w, x);
|
||||
x = ggml_add(ctx, x, b);
|
||||
return x;
|
||||
}
|
||||
|
||||
// w: [OC,IC, KH, KW]
|
||||
// x: [N, IC, IH, IW]
|
||||
// b: [OC,]
|
||||
// result: [N, OC, OH, OW]
|
||||
struct ggml_tensor* ggml_nn_conv_2d(struct ggml_context* ctx,
|
||||
struct ggml_tensor* x,
|
||||
struct ggml_tensor* w,
|
||||
struct ggml_tensor* b,
|
||||
int s0 = 1,
|
||||
int s1 = 1,
|
||||
int p0 = 0,
|
||||
int p1 = 0,
|
||||
int d0 = 1,
|
||||
int d1 = 1) {
|
||||
x = ggml_conv_2d(ctx, w, x, s0, s1, p0, p1, d0, d1);
|
||||
if (b != NULL) {
|
||||
b = ggml_reshape_4d(ctx, b, 1, 1, b->ne[0], 1);
|
||||
x = ggml_add(ctx, x, b);
|
||||
}
|
||||
return x;
|
||||
}
|
||||
|
||||
struct ggml_tensor* ggml_nn_layer_norm(struct ggml_context* ctx,
|
||||
struct ggml_tensor* x,
|
||||
struct ggml_tensor* w,
|
||||
struct ggml_tensor* b,
|
||||
float eps = EPS) {
|
||||
x = ggml_norm(ctx, x, eps);
|
||||
x = ggml_mul(ctx, x, w);
|
||||
x = ggml_add(ctx, x, b);
|
||||
return x;
|
||||
}
|
||||
|
||||
struct ggml_tensor* ggml_nn_group_norm(struct ggml_context* ctx,
|
||||
struct ggml_tensor* x,
|
||||
struct ggml_tensor* w,
|
||||
struct ggml_tensor* b,
|
||||
int num_groups = 32) {
|
||||
if (x->n_dims == 4) {
|
||||
w = ggml_reshape_4d(ctx, w, 1, 1, w->ne[0], 1);
|
||||
b = ggml_reshape_4d(ctx, b, 1, 1, b->ne[0], 1);
|
||||
}
|
||||
|
||||
x = ggml_group_norm(ctx, x, num_groups);
|
||||
x = ggml_mul(ctx, x, w);
|
||||
x = ggml_add(ctx, x, b);
|
||||
return x;
|
||||
}
|
||||
|
||||
std::pair<std::unordered_map<std::string, float>, std::string> extract_and_remove_lora(std::string text) {
|
||||
std::regex re("<lora:([^:]+):([^>]+)>");
|
||||
std::smatch matches;
|
||||
@@ -738,30 +807,21 @@ struct ResidualAttentionBlock {
|
||||
struct ggml_tensor* r = x;
|
||||
|
||||
// layer norm 1
|
||||
{
|
||||
x = ggml_norm(ctx, x, EPS);
|
||||
x = ggml_add(ctx,
|
||||
ggml_mul(ctx, x, ln1_w),
|
||||
ln1_b);
|
||||
}
|
||||
x = ggml_nn_layer_norm(ctx, x, ln1_w, ln1_b);
|
||||
// self-attention
|
||||
{
|
||||
struct ggml_tensor* q = ggml_add(ctx,
|
||||
ggml_mul_mat(ctx, q_w, x),
|
||||
q_b);
|
||||
struct ggml_tensor* q = ggml_nn_linear(ctx, x, q_w, q_b);
|
||||
q = ggml_scale_inplace(ctx, q, attn_scale);
|
||||
q = ggml_reshape_4d(ctx, q, d_model, n_head, n_token, N); // [N, n_token, n_head, d_model]
|
||||
q = ggml_cont(ctx, ggml_permute(ctx, q, 0, 2, 1, 3)); // [N, n_head, n_token, d_model]
|
||||
q = ggml_reshape_3d(ctx, q, d_model, n_token, n_head * N); // [N * n_head, n_token, d_model]
|
||||
|
||||
struct ggml_tensor* k = ggml_add(ctx,
|
||||
ggml_mul_mat(ctx, k_w, x), k_b);
|
||||
struct ggml_tensor* k = ggml_nn_linear(ctx, x, k_w, k_b);
|
||||
k = ggml_reshape_4d(ctx, k, d_model, n_head, n_token, N); // [N, n_token, n_head, d_model]
|
||||
k = ggml_cont(ctx, ggml_permute(ctx, k, 0, 2, 1, 3)); // [N, n_head, n_token, d_model]
|
||||
k = ggml_reshape_3d(ctx, k, d_model, n_token, n_head); // [N * n_head, n_token, d_model]
|
||||
|
||||
struct ggml_tensor* v = ggml_add(ctx,
|
||||
ggml_mul_mat(ctx, v_w, x), v_b);
|
||||
struct ggml_tensor* v = ggml_nn_linear(ctx, x, v_w, v_b);
|
||||
v = ggml_reshape_4d(ctx, v, d_model, n_head, n_token, N); // [N, n_token, n_head, d_model]
|
||||
v = ggml_cont(ctx, ggml_permute(ctx, v, 1, 2, 0, 3)); // [N, n_head, d_model, n_token]
|
||||
v = ggml_reshape_3d(ctx, v, n_token, d_model, n_head * N); // [N * n_head, d_model, n_token]
|
||||
@@ -779,24 +839,17 @@ struct ResidualAttentionBlock {
|
||||
}
|
||||
|
||||
// attention output
|
||||
x = ggml_mul_mat(ctx, out_w, x);
|
||||
x = ggml_add(ctx, x, out_b);
|
||||
x = ggml_nn_linear(ctx, x, out_w, out_b);
|
||||
|
||||
// residual
|
||||
x = ggml_add(ctx, x, r);
|
||||
r = x;
|
||||
|
||||
// layer norm 2
|
||||
{
|
||||
x = ggml_norm(ctx, x, EPS);
|
||||
|
||||
x = ggml_add(ctx, ggml_mul(ctx, x, ln2_w),
|
||||
ln2_b);
|
||||
}
|
||||
x = ggml_nn_layer_norm(ctx, x, ln2_w, ln2_b);
|
||||
|
||||
// mlp
|
||||
x = ggml_mul_mat(ctx, fc1_w, x);
|
||||
x = ggml_add(ctx, x, fc1_b);
|
||||
x = ggml_nn_linear(ctx, x, fc1_w, fc1_b);
|
||||
|
||||
if (hidden_size == 1024) { // SD 2.x
|
||||
x = ggml_gelu_inplace(ctx, x);
|
||||
@@ -804,8 +857,7 @@ struct ResidualAttentionBlock {
|
||||
x = ggml_gelu_quick_inplace(ctx, x);
|
||||
}
|
||||
|
||||
x = ggml_mul_mat(ctx, fc2_w, x);
|
||||
x = ggml_add(ctx, x, fc2_b);
|
||||
x = ggml_nn_linear(ctx, x, fc2_w, fc2_b);
|
||||
|
||||
// residual 2
|
||||
x = ggml_add(ctx, x, r);
|
||||
@@ -993,12 +1045,7 @@ struct CLIPTextModel {
|
||||
}
|
||||
|
||||
// final layer norm
|
||||
{
|
||||
x = ggml_norm(ctx0, x, EPS);
|
||||
|
||||
x = ggml_add(ctx0, ggml_mul(ctx0, x, final_ln_w),
|
||||
final_ln_b);
|
||||
}
|
||||
x = ggml_nn_layer_norm(ctx0, x, final_ln_w, final_ln_b);
|
||||
|
||||
return x; // [N, n_token, hidden_size]
|
||||
}
|
||||
@@ -1077,6 +1124,7 @@ struct CLIPTextModel {
|
||||
ggml_backend_buffer_free(compute_buffer);
|
||||
compute_alloc = NULL;
|
||||
compute_memory_buffer_size = -1;
|
||||
work_output = NULL;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -1252,48 +1300,29 @@ struct ResBlock {
|
||||
// emb: [N, emb_channels]
|
||||
|
||||
// in_layers
|
||||
// group norm 32
|
||||
auto h = ggml_group_norm_32(ctx, x);
|
||||
h = ggml_add(ctx,
|
||||
ggml_mul(ctx,
|
||||
h,
|
||||
ggml_reshape_4d(ctx, in_layer_0_w, 1, 1, in_layer_0_w->ne[0], 1)),
|
||||
ggml_reshape_4d(ctx, in_layer_0_b, 1, 1, in_layer_0_b->ne[0], 1));
|
||||
// silu
|
||||
h = ggml_silu_inplace(ctx, h);
|
||||
// conv2d
|
||||
h = ggml_conv_2d(ctx, in_layer_2_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx,
|
||||
h,
|
||||
ggml_reshape_4d(ctx, in_layer_2_b, 1, 1, in_layer_2_b->ne[0], 1)); // [N, out_channels, h, w]
|
||||
auto h = ggml_nn_group_norm(ctx, x, in_layer_0_w, in_layer_0_b);
|
||||
h = ggml_silu_inplace(ctx, h);
|
||||
h = ggml_nn_conv_2d(ctx, h, in_layer_2_w, in_layer_2_b, 1, 1, 1, 1); // [N, out_channels, h, w]
|
||||
|
||||
// emb_layers
|
||||
auto emb_out = ggml_silu(ctx, emb);
|
||||
emb_out = ggml_mul_mat(ctx, emb_layer_1_w, emb_out);
|
||||
emb_out = ggml_add(ctx, emb_out, emb_layer_1_b); // [N, out_channels]
|
||||
emb_out = ggml_nn_linear(ctx, emb_out, emb_layer_1_w, emb_layer_1_b); // [N, out_channels]
|
||||
emb_out = ggml_reshape_4d(ctx, emb_out, 1, 1, emb_out->ne[0], emb_out->ne[1]); // [N, out_channels, 1, 1]
|
||||
|
||||
// out_layers
|
||||
h = ggml_add(ctx, h, emb_out);
|
||||
// group norm 32
|
||||
h = ggml_group_norm_inplace(ctx, h, 32);
|
||||
h = ggml_add(ctx,
|
||||
ggml_mul(ctx, h, ggml_reshape_4d(ctx, out_layer_0_w, 1, 1, out_layer_0_w->ne[0], 1)),
|
||||
ggml_reshape_4d(ctx, out_layer_0_b, 1, 1, out_layer_0_b->ne[0], 1));
|
||||
// silu
|
||||
h = ggml_nn_group_norm(ctx, h, out_layer_0_w, out_layer_0_b);
|
||||
h = ggml_silu_inplace(ctx, h);
|
||||
|
||||
// dropout, skip for inference
|
||||
// conv2d
|
||||
h = ggml_conv_2d(ctx, out_layer_3_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx,
|
||||
h, ggml_reshape_4d(ctx, out_layer_3_b, 1, 1, out_layer_3_b->ne[0], 1)); // [N, out_channels, h, w
|
||||
|
||||
h = ggml_nn_conv_2d(ctx, h, out_layer_3_w, out_layer_3_b, 1, 1, 1, 1); // [N, out_channels, h, w]
|
||||
|
||||
// skip connection
|
||||
if (out_channels != channels) {
|
||||
x = ggml_conv_2d(ctx, skip_w, x, 1, 1, 0, 0, 1, 1);
|
||||
x = ggml_add(ctx,
|
||||
x, ggml_reshape_4d(ctx, skip_b, 1, 1, skip_b->ne[0], 1)); // [N, out_channels, h, w]
|
||||
x = ggml_nn_conv_2d(ctx, x, skip_w, skip_b); // [N, out_channels, h, w]
|
||||
}
|
||||
|
||||
h = ggml_add(ctx, h, x);
|
||||
return h; // [N, out_channels, h, w]
|
||||
}
|
||||
@@ -1468,15 +1497,9 @@ struct SpatialTransformer {
|
||||
// x: [N, in_channels, h, w]
|
||||
// context: [N, max_position, hidden_size(aka context_dim)]
|
||||
auto x_in = x;
|
||||
// group norm 32
|
||||
x = ggml_group_norm_32(ctx, x);
|
||||
x = ggml_add(ctx,
|
||||
ggml_mul(ctx, x, ggml_reshape_4d(ctx, norm_w, 1, 1, norm_w->ne[0], 1)),
|
||||
ggml_reshape_4d(ctx, norm_b, 1, 1, norm_b->ne[0], 1));
|
||||
x = ggml_nn_group_norm(ctx, x, norm_w, norm_b);
|
||||
// proj_in
|
||||
x = ggml_conv_2d(ctx, proj_in_w, x, 1, 1, 0, 0, 1, 1);
|
||||
x = ggml_add(ctx,
|
||||
x, ggml_reshape_4d(ctx, proj_in_b, 1, 1, proj_in_b->ne[0], 1)); // [N, in_channels, h, w]
|
||||
x = ggml_nn_conv_2d(ctx, x, proj_in_w, proj_in_b); // [N, in_channels, h, w]
|
||||
|
||||
// transformer
|
||||
const int64_t n = x->ne[3];
|
||||
@@ -1489,13 +1512,8 @@ struct SpatialTransformer {
|
||||
{
|
||||
auto r = x;
|
||||
// layer norm 1
|
||||
{
|
||||
x = ggml_reshape_2d(ctx, x, c, w * h * n);
|
||||
x = ggml_norm(ctx, x, EPS);
|
||||
x = ggml_add(ctx,
|
||||
ggml_mul(ctx, x, transformer.norm1_w),
|
||||
transformer.norm1_b);
|
||||
}
|
||||
x = ggml_reshape_2d(ctx, x, c, w * h * n);
|
||||
x = ggml_nn_layer_norm(ctx, x, transformer.norm1_w, transformer.norm1_b);
|
||||
|
||||
// self-attention
|
||||
{
|
||||
@@ -1533,7 +1551,7 @@ struct SpatialTransformer {
|
||||
// x = ggml_cpy(ctx, kqv, ggml_new_tensor_2d(ctx, GGML_TYPE_F32, d_head * n_head, h * w * n));
|
||||
x = ggml_reshape_2d(ctx, kqv, d_head * n_head, h * w * n);
|
||||
|
||||
x = ggml_add(ctx, ggml_mul_mat(ctx, transformer.attn1_out_w, x), transformer.attn1_out_b);
|
||||
x = ggml_nn_linear(ctx, x, transformer.attn1_out_w, transformer.attn1_out_b);
|
||||
|
||||
x = ggml_reshape_4d(ctx, x, c, w, h, n);
|
||||
}
|
||||
@@ -1542,11 +1560,7 @@ struct SpatialTransformer {
|
||||
r = x;
|
||||
|
||||
// layer norm 2
|
||||
{
|
||||
x = ggml_norm(ctx, x, EPS);
|
||||
x = ggml_add(ctx,
|
||||
ggml_mul(ctx, x, transformer.norm2_w), transformer.norm2_b);
|
||||
}
|
||||
x = ggml_nn_layer_norm(ctx, x, transformer.norm2_w, transformer.norm2_b);
|
||||
|
||||
// cross-attention
|
||||
{
|
||||
@@ -1584,7 +1598,7 @@ struct SpatialTransformer {
|
||||
// x = ggml_cpy(ctx, kqv, ggml_new_tensor_2d(ctx, GGML_TYPE_F32, d_head * n_head, h * w * n)); // [N * h * w, in_channels]
|
||||
x = ggml_reshape_2d(ctx, kqv, d_head * n_head, h * w * n); // [N * h * w, in_channels]
|
||||
|
||||
x = ggml_add(ctx, ggml_mul_mat(ctx, transformer.attn2_out_w, x), transformer.attn2_out_b);
|
||||
x = ggml_nn_linear(ctx, x, transformer.attn2_out_w, transformer.attn2_out_b);
|
||||
|
||||
x = ggml_reshape_4d(ctx, x, c, w, h, n);
|
||||
}
|
||||
@@ -1593,13 +1607,8 @@ struct SpatialTransformer {
|
||||
r = x;
|
||||
|
||||
// layer norm 3
|
||||
{
|
||||
x = ggml_reshape_2d(ctx, x, c, h * w * n); // [N * h * w, in_channels]
|
||||
x = ggml_norm(ctx, x, EPS);
|
||||
x = ggml_add(ctx,
|
||||
ggml_mul(ctx, x, transformer.norm3_w),
|
||||
transformer.norm3_b);
|
||||
}
|
||||
x = ggml_reshape_2d(ctx, x, c, h * w * n); // [N * h * w, in_channels]
|
||||
x = ggml_nn_layer_norm(ctx, x, transformer.norm3_w, transformer.norm3_b);
|
||||
|
||||
// ff
|
||||
{
|
||||
@@ -1626,17 +1635,14 @@ struct SpatialTransformer {
|
||||
transformer.ff_0_proj_b->nb[0] * transformer.ff_0_proj_b->ne[0] / 2); // [in_channels * 4, ]
|
||||
x = ggml_reshape_2d(ctx, x, c, w * h * n);
|
||||
auto x_in = x;
|
||||
x = ggml_mul_mat(ctx, x_w, x_in); // [N * h * w, in_channels * 4]
|
||||
x = ggml_add(ctx, x, x_b);
|
||||
auto gate = ggml_mul_mat(ctx, gate_w, x_in); // [N * h * w, in_channels * 4]
|
||||
gate = ggml_add(ctx, gate, gate_b);
|
||||
x = ggml_nn_linear(ctx, x_in, x_w, x_b); // [N * h * w, in_channels * 4]
|
||||
auto gate = ggml_nn_linear(ctx, x_in, gate_w, gate_b); // [N * h * w, in_channels * 4]
|
||||
|
||||
gate = ggml_gelu_inplace(ctx, gate);
|
||||
|
||||
x = ggml_mul(ctx, x, gate); // [N * h * w, in_channels * 4]
|
||||
// fc
|
||||
x = ggml_mul_mat(ctx, transformer.ff_2_w, x); // [N * h * w, in_channels]
|
||||
x = ggml_add(ctx, x, transformer.ff_2_b);
|
||||
x = ggml_nn_linear(ctx, x, transformer.ff_2_w, transformer.ff_2_b); // [N * h * w, in_channels]
|
||||
}
|
||||
|
||||
x = ggml_reshape_4d(ctx, x, c, w, h, n); // [N, h, w, in_channels]
|
||||
@@ -1644,12 +1650,11 @@ struct SpatialTransformer {
|
||||
// residual
|
||||
x = ggml_add(ctx, x, r);
|
||||
}
|
||||
x = ggml_cont(ctx, ggml_permute(ctx, x, 2, 0, 1, 3)); // // [N, in_channels, h, w]
|
||||
x = ggml_cont(ctx, ggml_permute(ctx, x, 2, 0, 1, 3)); // [N, in_channels, h, w]
|
||||
|
||||
// proj_out
|
||||
x = ggml_conv_2d(ctx, proj_out_w, x, 1, 1, 0, 0, 1, 1);
|
||||
x = ggml_add(ctx,
|
||||
x, ggml_reshape_4d(ctx, proj_out_b, 1, 1, proj_out_b->ne[0], 1)); // [N, in_channels, h, w]
|
||||
x = ggml_nn_conv_2d(ctx, x, proj_out_w, proj_out_b); // [N, in_channels, h, w]
|
||||
|
||||
x = ggml_add(ctx, x, x_in);
|
||||
return x;
|
||||
}
|
||||
@@ -1690,17 +1695,14 @@ struct DownSample {
|
||||
|
||||
struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* x) {
|
||||
// x: [N, channels, h, w]
|
||||
struct ggml_tensor* c = nullptr;
|
||||
struct ggml_tensor* c = NULL;
|
||||
if (vae_downsample) {
|
||||
c = ggml_pad(ctx, x, 1, 1, 0, 0);
|
||||
c = ggml_conv_2d(ctx, op_w, c, 2, 2, 0, 0, 1, 1);
|
||||
c = ggml_nn_conv_2d(ctx, c, op_w, op_b, 2, 2, 0, 0);
|
||||
} else {
|
||||
c = ggml_conv_2d(ctx, op_w, x, 2, 2, 1, 1, 1, 1);
|
||||
c = ggml_nn_conv_2d(ctx, x, op_w, op_b, 2, 2, 1, 1);
|
||||
}
|
||||
c = ggml_add(ctx,
|
||||
c,
|
||||
ggml_reshape_4d(ctx, op_b, 1, 1, op_b->ne[0], 1)); // [N, out_channels, h/2, w/2]
|
||||
return c;
|
||||
return c; // [N, out_channels, h/2, w/2]
|
||||
}
|
||||
};
|
||||
|
||||
@@ -1732,11 +1734,8 @@ struct UpSample {
|
||||
|
||||
struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* x) {
|
||||
// x: [N, channels, h, w]
|
||||
x = ggml_upscale(ctx, x, 2); // [N, channels, h*2, w*2]
|
||||
x = ggml_conv_2d(ctx, conv_w, x, 1, 1, 1, 1, 1, 1);
|
||||
x = ggml_add(ctx,
|
||||
x,
|
||||
ggml_reshape_4d(ctx, conv_b, 1, 1, conv_b->ne[0], 1)); // [N, out_channels, h*2, w*2]
|
||||
x = ggml_upscale(ctx, x, 2); // [N, channels, h*2, w*2]
|
||||
x = ggml_nn_conv_2d(ctx, x, conv_w, conv_b, 1, 1, 1, 1); // [N, out_channels, h*2, w*2]
|
||||
return x;
|
||||
}
|
||||
};
|
||||
@@ -2201,14 +2200,10 @@ struct UNetModel {
|
||||
}
|
||||
|
||||
// time_embed = nn.Sequential
|
||||
auto emb = ggml_nn_linear(ctx0, t_emb, time_embed_0_w, time_embed_0_b);
|
||||
emb = ggml_silu_inplace(ctx0, emb);
|
||||
// Linear
|
||||
auto emb = ggml_mul_mat(ctx0, time_embed_0_w, t_emb);
|
||||
emb = ggml_add(ctx0, emb, time_embed_0_b);
|
||||
// nn.SiLU()
|
||||
emb = ggml_silu_inplace(ctx0, emb);
|
||||
// Linear
|
||||
emb = ggml_mul_mat(ctx0, time_embed_2_w, emb);
|
||||
emb = ggml_add(ctx0, emb, time_embed_2_b); // [N, time_embed_dim]
|
||||
emb = ggml_nn_linear(ctx0, emb, time_embed_2_w, time_embed_2_b); // [N, time_embed_dim]
|
||||
|
||||
// SDXL
|
||||
// label_emd = nn.Sequential
|
||||
@@ -2216,13 +2211,9 @@ struct UNetModel {
|
||||
// param y: an [N] Tensor of labels, if class-conditional. (clip g)
|
||||
|
||||
// if(y != NULL) {
|
||||
// auto y_emb = ggml_mul_mat(ctx, label_embed_0_w, y);
|
||||
// y_emb = ggml_add(ctx, y_emb, label_embed_0_b);
|
||||
// // nn.SiLU()
|
||||
// auto y_emb = ggml_nn_linear(ctx, y, label_embed_0_w, label_embed_0_b);
|
||||
// y_emb = ggml_silu_inplace(ctx, y_emb);
|
||||
// // Linear
|
||||
// y_emb = ggml_mul_mat(ctx, label_embed_2_w, y_emb);
|
||||
// y_emb = ggml_add(ctx, y_emb, label_embed_2_b);
|
||||
// y_emb = ggml_nn_linear(ctx, y_emb, label_embed_2_w, label_embed_2_b);
|
||||
// emb = ggml_add(ctx, emb, y_emb);
|
||||
// }
|
||||
|
||||
@@ -2230,11 +2221,8 @@ struct UNetModel {
|
||||
std::vector<struct ggml_tensor*> hs;
|
||||
|
||||
// input block 0
|
||||
struct ggml_tensor* h = ggml_conv_2d(ctx0, input_block_0_w, x, 1, 1, 1, 1, 1, 1); // [N, model_channels, h, w]
|
||||
struct ggml_tensor* h = ggml_nn_conv_2d(ctx0, x, input_block_0_w, input_block_0_b, 1, 1, 1, 1); // [N, model_channels, h, w]
|
||||
|
||||
h = ggml_add(ctx0,
|
||||
h,
|
||||
ggml_reshape_4d(ctx0, input_block_0_b, 1, 1, input_block_0_b->ne[0], 1)); // [N, model_channels, h, w]
|
||||
ggml_set_name(h, "bench-start");
|
||||
hs.push_back(h);
|
||||
// input block 1-11
|
||||
@@ -2284,18 +2272,11 @@ struct UNetModel {
|
||||
}
|
||||
|
||||
// out
|
||||
// group norm 32
|
||||
h = ggml_group_norm_32(ctx0, h);
|
||||
h = ggml_add(ctx0,
|
||||
ggml_mul(ctx0, h, ggml_reshape_4d(ctx0, out_0_w, 1, 1, out_0_w->ne[0], 1)),
|
||||
ggml_reshape_4d(ctx0, out_0_b, 1, 1, out_0_b->ne[0], 1));
|
||||
// silu
|
||||
h = ggml_nn_group_norm(ctx0, h, out_0_w, out_0_b);
|
||||
h = ggml_silu_inplace(ctx0, h);
|
||||
|
||||
// conv2d
|
||||
h = ggml_conv_2d(ctx0, out_2_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx0,
|
||||
h, ggml_reshape_4d(ctx0, out_2_b, 1, 1, out_2_b->ne[0], 1)); // [N, out_channels, h, w]
|
||||
h = ggml_nn_conv_2d(ctx0, h, out_2_w, out_2_b, 1, 1, 1, 1); // [N, out_channels, h, w]
|
||||
ggml_set_name(h, "bench-end");
|
||||
return h;
|
||||
}
|
||||
@@ -2492,38 +2473,19 @@ struct ResnetBlock {
|
||||
struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* z) {
|
||||
// z: [N, in_channels, h, w]
|
||||
|
||||
// group norm 32
|
||||
auto h = ggml_group_norm_32(ctx, z);
|
||||
h = ggml_mul(ctx,
|
||||
h, ggml_reshape_4d(ctx, norm1_w, 1, 1, norm1_w->ne[0], 1));
|
||||
h = ggml_add(ctx,
|
||||
h, ggml_reshape_4d(ctx, norm1_b, 1, 1, norm1_b->ne[0], 1));
|
||||
// silu
|
||||
h = ggml_silu_inplace(ctx, h);
|
||||
// conv2d
|
||||
h = ggml_conv_2d(ctx, conv1_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx,
|
||||
h, ggml_reshape_4d(ctx, conv1_b, 1, 1, conv1_b->ne[0], 1)); // [N, out_channels, h, w]
|
||||
|
||||
// group norm 32
|
||||
h = ggml_group_norm_32(ctx, h);
|
||||
h = ggml_add(ctx,
|
||||
ggml_mul(ctx, h, ggml_reshape_4d(ctx, norm2_w, 1, 1, norm2_w->ne[0], 1)),
|
||||
ggml_reshape_4d(ctx, norm2_b, 1, 1, norm2_b->ne[0], 1));
|
||||
// silu
|
||||
h = ggml_silu_inplace(ctx, h);
|
||||
auto h = ggml_nn_group_norm(ctx, z, norm1_w, norm1_b);
|
||||
h = ggml_silu_inplace(ctx, h);
|
||||
h = ggml_nn_conv_2d(ctx, h, conv1_w, conv1_b, 1, 1, 1, 1); // [N, out_channels, h, w]
|
||||
h = ggml_nn_group_norm(ctx, h, norm2_w, norm2_b);
|
||||
h = ggml_silu_inplace(ctx, h);
|
||||
// dropout, skip for inference
|
||||
// conv2d
|
||||
h = ggml_conv_2d(ctx, conv2_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx,
|
||||
h, ggml_reshape_4d(ctx, conv2_b, 1, 1, conv2_b->ne[0], 1)); // [N, out_channels, h, w
|
||||
h = ggml_nn_conv_2d(ctx, h, conv2_w, conv2_b, 1, 1, 1, 1); // [N, out_channels, h, w]
|
||||
|
||||
// skip connection
|
||||
if (out_channels != in_channels) {
|
||||
z = ggml_conv_2d(ctx, nin_shortcut_w, z, 1, 1, 0, 0, 1, 1);
|
||||
z = ggml_add(ctx,
|
||||
z, ggml_reshape_4d(ctx, nin_shortcut_b, 1, 1, nin_shortcut_b->ne[0], 1)); // [N, out_channels, h, w]
|
||||
z = ggml_nn_conv_2d(ctx, z, nin_shortcut_w, nin_shortcut_b); // [N, out_channels, h, w]
|
||||
}
|
||||
|
||||
h = ggml_add(ctx, h, z);
|
||||
return h; // [N, out_channels, h, w]
|
||||
}
|
||||
@@ -2593,30 +2555,16 @@ struct AttnBlock {
|
||||
struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* x) {
|
||||
// x: [N, in_channels, h, w]
|
||||
|
||||
// group norm 32
|
||||
auto h_ = ggml_group_norm_32(ctx, x);
|
||||
h_ = ggml_add(ctx,
|
||||
ggml_mul(ctx, h_, ggml_reshape_4d(ctx, norm_w, 1, 1, norm_w->ne[0], 1)),
|
||||
ggml_reshape_4d(ctx, norm_b, 1, 1, norm_b->ne[0], 1));
|
||||
auto h_ = ggml_nn_group_norm(ctx, x, norm_w, norm_b);
|
||||
|
||||
const int64_t n = h_->ne[3];
|
||||
const int64_t c = h_->ne[2];
|
||||
const int64_t h = h_->ne[1];
|
||||
const int64_t w = h_->ne[0];
|
||||
// q
|
||||
auto q = ggml_conv_2d(ctx, q_w, h_, 1, 1, 0, 0, 1, 1);
|
||||
q = ggml_add(ctx,
|
||||
q, ggml_reshape_4d(ctx, q_b, 1, 1, q_b->ne[0], 1)); // [N, in_channels, h, w]
|
||||
|
||||
// k
|
||||
auto k = ggml_conv_2d(ctx, k_w, h_, 1, 1, 0, 0, 1, 1);
|
||||
k = ggml_add(ctx,
|
||||
k, ggml_reshape_4d(ctx, k_b, 1, 1, k_b->ne[0], 1)); // [N, in_channels, h, w]
|
||||
|
||||
// v
|
||||
auto v = ggml_conv_2d(ctx, v_w, h_, 1, 1, 0, 0, 1, 1);
|
||||
v = ggml_add(ctx,
|
||||
v, ggml_reshape_4d(ctx, v_b, 1, 1, v_b->ne[0], 1)); // [N, in_channels, h, w]
|
||||
auto q = ggml_nn_conv_2d(ctx, h_, q_w, q_b); // [N, in_channels, h, w]
|
||||
auto k = ggml_nn_conv_2d(ctx, h_, k_w, k_b); // [N, in_channels, h, w]
|
||||
auto v = ggml_nn_conv_2d(ctx, h_, v_w, v_b); // [N, in_channels, h, w]
|
||||
|
||||
q = ggml_cont(ctx, ggml_permute(ctx, q, 1, 2, 0, 3)); // [N, h, w, in_channels]
|
||||
q = ggml_reshape_3d(ctx, q, c, h * w, n); // [N, h * w, in_channels]
|
||||
@@ -2634,9 +2582,8 @@ struct AttnBlock {
|
||||
h_ = ggml_reshape_4d(ctx, h_, w, h, c, n); // [N, in_channels, h, w]
|
||||
|
||||
// proj_out
|
||||
h_ = ggml_conv_2d(ctx, proj_out_w, h_, 1, 1, 0, 0, 1, 1);
|
||||
h_ = ggml_add(ctx,
|
||||
h_, ggml_reshape_4d(ctx, proj_out_b, 1, 1, proj_out_b->ne[0], 1)); // [N, in_channels, h, w]
|
||||
h_ = ggml_nn_conv_2d(ctx, h_, proj_out_w, proj_out_b); // [N, in_channels, h, w]
|
||||
|
||||
h_ = ggml_add(ctx, h_, x);
|
||||
return h_;
|
||||
}
|
||||
@@ -2803,9 +2750,7 @@ struct Encoder {
|
||||
// x: [N, in_channels, h, w]
|
||||
|
||||
// conv_in
|
||||
auto h = ggml_conv_2d(ctx, conv_in_w, x, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx,
|
||||
h, ggml_reshape_4d(ctx, conv_in_b, 1, 1, conv_in_b->ne[0], 1)); // [N, ch, h, w]
|
||||
auto h = ggml_nn_conv_2d(ctx, x, conv_in_w, conv_in_b, 1, 1, 1, 1); // [N, ch, h, w]
|
||||
ggml_set_name(h, "b-start");
|
||||
int len_mults = sizeof(ch_mult) / sizeof(int);
|
||||
for (int i = 0; i < len_mults; i++) {
|
||||
@@ -2821,20 +2766,11 @@ struct Encoder {
|
||||
h = mid.attn_1.forward(ctx, h);
|
||||
h = mid.block_2.forward(ctx, h); // [N, block_in, h, w]
|
||||
|
||||
// group norm 32
|
||||
h = ggml_group_norm_32(ctx, h);
|
||||
h = ggml_add(ctx,
|
||||
ggml_mul(ctx, h, ggml_reshape_4d(ctx, norm_out_w, 1, 1, norm_out_w->ne[0], 1)),
|
||||
ggml_reshape_4d(ctx, norm_out_b, 1, 1, norm_out_b->ne[0], 1));
|
||||
|
||||
// silu
|
||||
// silu
|
||||
h = ggml_nn_group_norm(ctx, h, norm_out_w, norm_out_b);
|
||||
h = ggml_silu_inplace(ctx, h);
|
||||
|
||||
// conv_out
|
||||
h = ggml_conv_2d(ctx, conv_out_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx,
|
||||
h, ggml_reshape_4d(ctx, conv_out_b, 1, 1, conv_out_b->ne[0], 1)); // [N, z_channels*2, h, w]
|
||||
h = ggml_nn_conv_2d(ctx, h, conv_out_w, conv_out_b, 1, 1, 1, 1); // [N, z_channels*2, h, w]
|
||||
|
||||
return h;
|
||||
}
|
||||
@@ -2996,9 +2932,7 @@ struct Decoder {
|
||||
struct ggml_tensor* forward(struct ggml_context* ctx, struct ggml_tensor* z) {
|
||||
// z: [N, z_channels, h, w]
|
||||
// conv_in
|
||||
auto h = ggml_conv_2d(ctx, conv_in_w, z, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx,
|
||||
h, ggml_reshape_4d(ctx, conv_in_b, 1, 1, conv_in_b->ne[0], 1)); // [N, block_in, h, w]
|
||||
auto h = ggml_nn_conv_2d(ctx, z, conv_in_w, conv_in_b, 1, 1, 1, 1); // [N, block_in, h, w]
|
||||
|
||||
h = mid.block_1.forward(ctx, h);
|
||||
h = mid.attn_1.forward(ctx, h);
|
||||
@@ -3015,19 +2949,11 @@ struct Decoder {
|
||||
}
|
||||
|
||||
// group norm 32
|
||||
h = ggml_group_norm_32(ctx, h);
|
||||
h = ggml_add(ctx,
|
||||
ggml_mul(ctx, h, ggml_reshape_4d(ctx, norm_out_w, 1, 1, norm_out_w->ne[0], 1)),
|
||||
ggml_reshape_4d(ctx, norm_out_b, 1, 1, norm_out_b->ne[0], 1));
|
||||
|
||||
// silu
|
||||
// silu
|
||||
h = ggml_nn_group_norm(ctx, h, norm_out_w, norm_out_b);
|
||||
h = ggml_silu_inplace(ctx, h);
|
||||
|
||||
// conv_out
|
||||
h = ggml_conv_2d(ctx, conv_out_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx,
|
||||
h, ggml_reshape_4d(ctx, conv_out_b, 1, 1, conv_out_b->ne[0], 1)); // [N, out_ch, h, w]
|
||||
h = ggml_nn_conv_2d(ctx, h, conv_out_w, conv_out_b, 1, 1, 1, 1); // [N, out_ch, h, w]
|
||||
return h;
|
||||
}
|
||||
};
|
||||
@@ -3176,9 +3102,7 @@ struct AutoEncoderKL {
|
||||
struct ggml_tensor* decode(struct ggml_context* ctx0, struct ggml_tensor* z) {
|
||||
// z: [N, z_channels, h, w]
|
||||
// post_quant_conv
|
||||
auto h = ggml_conv_2d(ctx0, post_quant_conv_w, z, 1, 1, 0, 0, 1, 1);
|
||||
h = ggml_add(ctx0,
|
||||
h, ggml_reshape_4d(ctx0, post_quant_conv_b, 1, 1, post_quant_conv_b->ne[0], 1)); // [N, z_channels, h, w]
|
||||
auto h = ggml_nn_conv_2d(ctx0, z, post_quant_conv_w, post_quant_conv_b); // [N, z_channels, h, w]
|
||||
ggml_set_name(h, "bench-start");
|
||||
h = decoder.forward(ctx0, h);
|
||||
ggml_set_name(h, "bench-end");
|
||||
@@ -3189,10 +3113,7 @@ struct AutoEncoderKL {
|
||||
// x: [N, in_channels, h, w]
|
||||
auto h = encoder.forward(ctx0, x); // [N, 2*z_channels, h/8, w/8]
|
||||
// quant_conv
|
||||
h = ggml_conv_2d(ctx0, quant_conv_w, h, 1, 1, 0, 0, 1, 1);
|
||||
h = ggml_add(ctx0,
|
||||
h,
|
||||
ggml_reshape_4d(ctx0, quant_conv_b, 1, 1, quant_conv_b->ne[0], 1)); // [N, 2*embed_dim, h/8, w/8]
|
||||
h = ggml_nn_conv_2d(ctx0, h, quant_conv_w, quant_conv_b); // [N, 2*embed_dim, h/8, w/8]
|
||||
ggml_set_name(h, "b-end");
|
||||
return h;
|
||||
}
|
||||
@@ -3356,25 +3277,16 @@ struct TAEBlock {
|
||||
ggml_tensor* forward(ggml_context* ctx, ggml_tensor* x) {
|
||||
// conv(n_in, n_out)
|
||||
ggml_tensor* h;
|
||||
h = ggml_conv_2d(ctx, conv_0_w, x, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx, h, ggml_reshape_4d(ctx, conv_0_b, 1, 1, conv_0_b->ne[0], 1));
|
||||
|
||||
// relu
|
||||
h = ggml_nn_conv_2d(ctx, x, conv_0_w, conv_0_b, 1, 1, 1, 1);
|
||||
h = ggml_relu_inplace(ctx, h);
|
||||
|
||||
h = ggml_conv_2d(ctx, conv_1_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx, h, ggml_reshape_4d(ctx, conv_1_b, 1, 1, conv_1_b->ne[0], 1));
|
||||
|
||||
// relu
|
||||
h = ggml_nn_conv_2d(ctx, h, conv_1_w, conv_1_b, 1, 1, 1, 1);
|
||||
h = ggml_relu_inplace(ctx, h);
|
||||
|
||||
h = ggml_conv_2d(ctx, conv_2_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx, h, ggml_reshape_4d(ctx, conv_2_b, 1, 1, conv_2_b->ne[0], 1));
|
||||
h = ggml_nn_conv_2d(ctx, h, conv_2_w, conv_2_b, 1, 1, 1, 1);
|
||||
|
||||
// skip connection
|
||||
if (in_channels != out_channels) {
|
||||
// skip = nn.Conv2d(n_in, n_out, 1, bias=False) if n_in != n_out else nn.Identity()
|
||||
x = ggml_conv_2d(ctx, conv_skip_w, x, 1, 1, 1, 1, 1, 1);
|
||||
x = ggml_nn_conv_2d(ctx, x, conv_skip_w, NULL, 1, 1, 1, 1);
|
||||
}
|
||||
|
||||
h = ggml_add(ctx, h, x);
|
||||
@@ -3503,14 +3415,13 @@ struct TinyEncoder {
|
||||
|
||||
ggml_tensor* forward(ggml_context* ctx, ggml_tensor* x) {
|
||||
// conv(3, 64)
|
||||
auto z = ggml_conv_2d(ctx, conv_input_w, x, 1, 1, 1, 1, 1, 1);
|
||||
z = ggml_add(ctx, z, ggml_reshape_4d(ctx, conv_input_b, 1, 1, conv_input_b->ne[0], 1));
|
||||
auto z = ggml_nn_conv_2d(ctx, x, conv_input_w, conv_input_b, 1, 1, 1, 1);
|
||||
|
||||
// Block(64, 64)
|
||||
z = initial_block.forward(ctx, z);
|
||||
|
||||
// conv(64, 64, stride=2, bias=False)
|
||||
z = ggml_conv_2d(ctx, conv_1_w, z, 2, 2, 1, 1, 1, 1);
|
||||
z = ggml_nn_conv_2d(ctx, z, conv_1_w, NULL, 2, 2, 1, 1);
|
||||
|
||||
// Block(64, 64), Block(64, 64), Block(64, 64)
|
||||
for (int i = 0; i < num_blocks; i++) {
|
||||
@@ -3518,7 +3429,7 @@ struct TinyEncoder {
|
||||
}
|
||||
|
||||
// conv(64, 64, stride=2, bias=False)
|
||||
z = ggml_conv_2d(ctx, conv_2_w, z, 2, 2, 1, 1, 1, 1);
|
||||
z = ggml_nn_conv_2d(ctx, z, conv_2_w, NULL, 2, 2, 1, 1);
|
||||
|
||||
// Block(64, 64), Block(64, 64), Block(64, 64)
|
||||
for (int i = 0; i < num_blocks; i++) {
|
||||
@@ -3526,7 +3437,7 @@ struct TinyEncoder {
|
||||
}
|
||||
|
||||
// conv(64, 64, stride=2, bias=False)
|
||||
z = ggml_conv_2d(ctx, conv_3_w, z, 2, 2, 1, 1, 1, 1);
|
||||
z = ggml_nn_conv_2d(ctx, z, conv_3_w, NULL, 2, 2, 1, 1);
|
||||
|
||||
// Block(64, 64), Block(64, 64), Block(64, 64)
|
||||
for (int i = 0; i < num_blocks; i++) {
|
||||
@@ -3534,8 +3445,7 @@ struct TinyEncoder {
|
||||
}
|
||||
|
||||
// conv(64, 4)
|
||||
z = ggml_conv_2d(ctx, conv_final_w, z, 1, 1, 1, 1, 1, 1);
|
||||
z = ggml_add(ctx, z, ggml_reshape_4d(ctx, conv_final_b, 1, 1, conv_final_b->ne[0], 1));
|
||||
z = ggml_nn_conv_2d(ctx, z, conv_final_w, conv_final_b, 1, 1, 1, 1);
|
||||
return z;
|
||||
}
|
||||
};
|
||||
@@ -3683,8 +3593,7 @@ struct TinyDecoder {
|
||||
h = ggml_scale(ctx, h, in_scale_3);
|
||||
|
||||
// conv(4, 64)
|
||||
h = ggml_conv_2d(ctx, conv_input_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx, h, ggml_reshape_4d(ctx, conv_input_b, 1, 1, conv_input_b->ne[0], 1));
|
||||
h = ggml_nn_conv_2d(ctx, h, conv_input_w, conv_input_b, 1, 1, 1, 1);
|
||||
|
||||
// nn.ReLU()
|
||||
h = ggml_relu_inplace(ctx, h);
|
||||
@@ -3698,7 +3607,7 @@ struct TinyDecoder {
|
||||
h = ggml_upscale(ctx, h, 2);
|
||||
|
||||
// conv(64, 64, bias=False)
|
||||
h = ggml_conv_2d(ctx, conv_1_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_nn_conv_2d(ctx, h, conv_1_w, NULL, 1, 1, 1, 1);
|
||||
|
||||
// Block(64, 64), Block(64, 64), Block(64, 64)
|
||||
for (int i = 0; i < num_blocks; i++) {
|
||||
@@ -3709,7 +3618,7 @@ struct TinyDecoder {
|
||||
h = ggml_upscale(ctx, h, 2);
|
||||
|
||||
// conv(64, 64, bias=False)
|
||||
h = ggml_conv_2d(ctx, conv_2_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_nn_conv_2d(ctx, h, conv_2_w, NULL, 1, 1, 1, 1);
|
||||
|
||||
// Block(64, 64), Block(64, 64), Block(64, 64)
|
||||
for (int i = 0; i < num_blocks; i++) {
|
||||
@@ -3720,14 +3629,13 @@ struct TinyDecoder {
|
||||
h = ggml_upscale(ctx, h, 2);
|
||||
|
||||
// conv(64, 64, bias=False)
|
||||
h = ggml_conv_2d(ctx, conv_3_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_nn_conv_2d(ctx, h, conv_3_w, NULL, 1, 1, 1, 1);
|
||||
|
||||
// Block(64, 64)
|
||||
h = final_block.forward(ctx, h);
|
||||
|
||||
// conv(64, 3)
|
||||
h = ggml_conv_2d(ctx, conv_final_w, h, 1, 1, 1, 1, 1, 1);
|
||||
h = ggml_add(ctx, h, ggml_reshape_4d(ctx, conv_final_b, 1, 1, conv_final_b->ne[0], 1));
|
||||
h = ggml_nn_conv_2d(ctx, h, conv_final_w, conv_final_b, 1, 1, 1, 1);
|
||||
return h;
|
||||
}
|
||||
};
|
||||
@@ -4724,7 +4632,7 @@ public:
|
||||
LOG_DEBUG("computing condition graph completed, taking %" PRId64 " ms", t1 - t0);
|
||||
ggml_tensor* result = ggml_dup_tensor(work_ctx, hidden_states);
|
||||
{
|
||||
float original_mean = sd_mean(hidden_states);
|
||||
float original_mean = ggml_tensor_mean(hidden_states);
|
||||
for (int i2 = 0; i2 < hidden_states->ne[2]; i2++) {
|
||||
for (int i1 = 0; i1 < hidden_states->ne[1]; i1++) {
|
||||
for (int i0 = 0; i0 < hidden_states->ne[0]; i0++) {
|
||||
@@ -4734,16 +4642,17 @@ public:
|
||||
}
|
||||
}
|
||||
}
|
||||
float new_mean = sd_mean(result);
|
||||
sd_scale(result, (original_mean / new_mean));
|
||||
float new_mean = ggml_tensor_mean(result);
|
||||
ggml_tensor_scale(result, (original_mean / new_mean));
|
||||
}
|
||||
return result; // [1, 77, 768]
|
||||
}
|
||||
|
||||
ggml_tensor* sample(ggml_context* work_ctx,
|
||||
ggml_tensor* x_t,
|
||||
ggml_tensor* positive,
|
||||
ggml_tensor* negative,
|
||||
ggml_tensor* noise,
|
||||
ggml_tensor* c,
|
||||
ggml_tensor* uc,
|
||||
float cfg_scale,
|
||||
SampleMethod method,
|
||||
const std::vector<float>& sigmas) {
|
||||
@@ -4756,12 +4665,18 @@ public:
|
||||
struct ggml_tensor* noised_input = ggml_dup_tensor(work_ctx, x_t);
|
||||
struct ggml_tensor* timesteps = ggml_new_tensor_1d(work_ctx, GGML_TYPE_F32, 1); // [N, ]
|
||||
struct ggml_tensor* t_emb = new_timestep_embedding(work_ctx, NULL, timesteps, diffusion_model.model_channels); // [N, model_channels]
|
||||
diffusion_model.begin(noised_input, positive, t_emb);
|
||||
diffusion_model.begin(noised_input, c, t_emb);
|
||||
|
||||
bool has_unconditioned = cfg_scale != 1.0 && negative != NULL;
|
||||
bool has_unconditioned = cfg_scale != 1.0 && uc != NULL;
|
||||
|
||||
// x = x * sigmas[0]
|
||||
sd_scale(x, sigmas[0]);
|
||||
if (noise == NULL) {
|
||||
// x = x * sigmas[0]
|
||||
ggml_tensor_scale(x, sigmas[0]);
|
||||
} else {
|
||||
// xi = x + noise * sigma_sched[0]
|
||||
ggml_tensor_scale(noise, sigmas[0]);
|
||||
ggml_tensor_add(x, noise);
|
||||
}
|
||||
|
||||
// denoise wrapper
|
||||
struct ggml_tensor* out_cond = ggml_dup_tensor(work_ctx, x);
|
||||
@@ -4797,15 +4712,15 @@ public:
|
||||
|
||||
copy_ggml_tensor(noised_input, input);
|
||||
// noised_input = noised_input * c_in
|
||||
sd_scale(noised_input, c_in);
|
||||
ggml_tensor_scale(noised_input, c_in);
|
||||
|
||||
// cond
|
||||
diffusion_model.compute(out_cond, n_threads, noised_input, NULL, positive, t_emb);
|
||||
diffusion_model.compute(out_cond, n_threads, noised_input, NULL, c, t_emb);
|
||||
|
||||
float* negative_data = NULL;
|
||||
if (has_unconditioned) {
|
||||
// uncond
|
||||
diffusion_model.compute(out_uncond, n_threads, noised_input, NULL, negative, t_emb);
|
||||
diffusion_model.compute(out_uncond, n_threads, noised_input, NULL, uc, t_emb);
|
||||
negative_data = (float*)out_uncond->data;
|
||||
}
|
||||
float* vec_denoised = (float*)denoised->data;
|
||||
@@ -5260,15 +5175,15 @@ public:
|
||||
int64_t t0 = ggml_time_ms();
|
||||
if (!use_tiny_autoencoder) {
|
||||
if (decode) {
|
||||
sd_scale(x, 1.0f / scale_factor);
|
||||
ggml_tensor_scale(x, 1.0f / scale_factor);
|
||||
} else {
|
||||
sd_convert_input(x);
|
||||
ggml_tensor_scale_input(x);
|
||||
}
|
||||
first_stage_model.begin(x, decode);
|
||||
first_stage_model.compute(result, n_threads, x, decode);
|
||||
first_stage_model.end();
|
||||
if (decode) {
|
||||
sd_convert_output(result);
|
||||
ggml_tensor_scale_output(result);
|
||||
}
|
||||
} else {
|
||||
tae_first_stage.begin(x, decode);
|
||||
@@ -5278,10 +5193,18 @@ public:
|
||||
int64_t t1 = ggml_time_ms();
|
||||
LOG_DEBUG("computing vae [mode: %s] graph completed, taking %.2fs", decode ? "DECODE" : "ENCODE", (t1 - t0) * 1.0f / 1000);
|
||||
if (decode) {
|
||||
sd_clamp(result, 0.0f, 1.0f);
|
||||
ggml_tensor_clamp(result, 0.0f, 1.0f);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
ggml_tensor* encode_first_stage(ggml_context* work_ctx, ggml_tensor* x) {
|
||||
return compute_first_stage(work_ctx, x, false);
|
||||
}
|
||||
|
||||
ggml_tensor* decode_first_stage(ggml_context* work_ctx, ggml_tensor* x) {
|
||||
return compute_first_stage(work_ctx, x, true);
|
||||
}
|
||||
};
|
||||
|
||||
/*================================================= StableDiffusion ==================================================*/
|
||||
@@ -5358,11 +5281,11 @@ std::vector<uint8_t*> StableDiffusion::txt2img(std::string prompt,
|
||||
seed = rand();
|
||||
}
|
||||
|
||||
t0 = ggml_time_ms();
|
||||
ggml_tensor* postive = sd->get_learned_condition(work_ctx, prompt);
|
||||
struct ggml_tensor* negative = NULL;
|
||||
t0 = ggml_time_ms();
|
||||
ggml_tensor* c = sd->get_learned_condition(work_ctx, prompt);
|
||||
struct ggml_tensor* uc = NULL;
|
||||
if (cfg_scale != 1.0) {
|
||||
negative = sd->get_learned_condition(work_ctx, negative_prompt);
|
||||
uc = sd->get_learned_condition(work_ctx, negative_prompt);
|
||||
}
|
||||
t1 = ggml_time_ms();
|
||||
LOG_INFO("get_learned_condition completed, taking %" PRId64 " ms", t1 - t0);
|
||||
@@ -5387,7 +5310,7 @@ std::vector<uint8_t*> StableDiffusion::txt2img(std::string prompt,
|
||||
|
||||
std::vector<float> sigmas = sd->denoiser->schedule->get_sigmas(sample_steps);
|
||||
|
||||
struct ggml_tensor* x_0 = sd->sample(work_ctx, x_t, postive, negative, cfg_scale, sample_method, sigmas);
|
||||
struct ggml_tensor* x_0 = sd->sample(work_ctx, x_t, NULL, c, uc, cfg_scale, sample_method, sigmas);
|
||||
// struct ggml_tensor* x_0 = load_tensor_from_file(ctx, "samples_ddim.bin");
|
||||
// print_ggml_tensor(x_0);
|
||||
int64_t sampling_end = ggml_time_ms();
|
||||
@@ -5404,7 +5327,7 @@ std::vector<uint8_t*> StableDiffusion::txt2img(std::string prompt,
|
||||
LOG_INFO("decoding %zu latents", final_latents.size());
|
||||
for (size_t i = 0; i < final_latents.size(); i++) {
|
||||
t1 = ggml_time_ms();
|
||||
struct ggml_tensor* img = sd->compute_first_stage(work_ctx, final_latents[i] /* x_0 */, true);
|
||||
struct ggml_tensor* img = sd->decode_first_stage(work_ctx, final_latents[i] /* x_0 */);
|
||||
if (img != NULL) {
|
||||
results.push_back(sd_tensor_to_image(img));
|
||||
}
|
||||
@@ -5483,10 +5406,10 @@ std::vector<uint8_t*> StableDiffusion::img2img(const uint8_t* init_img_data,
|
||||
t0 = ggml_time_ms();
|
||||
ggml_tensor* init_latent = NULL;
|
||||
if (!sd->use_tiny_autoencoder) {
|
||||
ggml_tensor* moments = sd->compute_first_stage(work_ctx, init_img, false);
|
||||
ggml_tensor* moments = sd->encode_first_stage(work_ctx, init_img);
|
||||
init_latent = sd->get_first_stage_encoding(work_ctx, moments);
|
||||
} else {
|
||||
init_latent = sd->compute_first_stage(work_ctx, init_img, false);
|
||||
init_latent = sd->encode_first_stage(work_ctx, init_img);
|
||||
}
|
||||
// print_ggml_tensor(init_latent);
|
||||
t1 = ggml_time_ms();
|
||||
@@ -5507,8 +5430,12 @@ std::vector<uint8_t*> StableDiffusion::img2img(const uint8_t* init_img_data,
|
||||
// requires encode_adm
|
||||
// apply set_timestep_embedding with dim 256
|
||||
|
||||
sd->rng->manual_seed(seed);
|
||||
struct ggml_tensor* noise = ggml_dup_tensor(work_ctx, init_latent);
|
||||
ggml_tensor_set_f32_randn(noise, sd->rng);
|
||||
|
||||
LOG_INFO("sampling using %s method", sampling_methods_str[sample_method]);
|
||||
struct ggml_tensor* x_0 = sd->sample(work_ctx, init_latent, c, uc, cfg_scale, sample_method, sigma_sched);
|
||||
struct ggml_tensor* x_0 = sd->sample(work_ctx, init_latent, noise, c, uc, cfg_scale, sample_method, sigma_sched);
|
||||
// struct ggml_tensor *x_0 = load_tensor_from_file(ctx, "samples_ddim.bin");
|
||||
// print_ggml_tensor(x_0);
|
||||
int64_t t3 = ggml_time_ms();
|
||||
@@ -5517,7 +5444,7 @@ std::vector<uint8_t*> StableDiffusion::img2img(const uint8_t* init_img_data,
|
||||
sd->diffusion_model.destroy();
|
||||
}
|
||||
|
||||
struct ggml_tensor* img = sd->compute_first_stage(work_ctx, x_0, true);
|
||||
struct ggml_tensor* img = sd->decode_first_stage(work_ctx, x_0);
|
||||
if (img != NULL) {
|
||||
result.push_back(sd_tensor_to_image(img));
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user