modify ggml core to support tts

This commit is contained in:
Concedo 2025-08-16 16:52:34 +08:00
parent d876898476
commit 2bf128587d
3 changed files with 1117 additions and 0 deletions

View file

@ -553,6 +553,18 @@ extern "C" {
GGML_OP_GLU,
GGML_OP_COUNT,
//kcpp: dirtypatch of unofficial ops for ttscpp
GGML_OP_UPSCALE_LINEAR, // linear interpolate
GGML_OP_RECIPROCAL,
GGML_OP_ROUND,
GGML_OP_MOD,
GGML_OP_CUMSUM,
GGML_OP_STFT,
GGML_OP_AA_STFT,
GGML_OP_ISTFT,
GGML_OP_AA_ISTFT,
GGML_OP_CONV_TRANSPOSE_1D_TTS,
};
enum ggml_unary_op {
@ -2475,6 +2487,76 @@ extern "C" {
GGML_API void ggml_threadpool_params_init (struct ggml_threadpool_params * p, int n_threads);
GGML_API bool ggml_threadpool_params_match (const struct ggml_threadpool_params * p0, const struct ggml_threadpool_params * p1);
//kcpp: dirtypatch of ttscpp additions
GGML_API struct ggml_tensor * ggml_round(
struct ggml_context * ctx,
struct ggml_tensor * a);
GGML_API struct ggml_tensor * ggml_round_inplace(
struct ggml_context * ctx,
struct ggml_tensor * a);
// This is a floating point mod by the mod_val parameter
GGML_API struct ggml_tensor * ggml_mod(
struct ggml_context * ctx,
struct ggml_tensor * a,
float mod_val);
GGML_API struct ggml_tensor * ggml_mod_inplace(
struct ggml_context * ctx,
struct ggml_tensor * a,
float mod_val);
// reciprocal of each value in tensor a
GGML_API struct ggml_tensor * ggml_reciprocal(
struct ggml_context * ctx,
struct ggml_tensor * a);
GGML_API struct ggml_tensor * ggml_reciprocal_inplace(
struct ggml_context * ctx,
struct ggml_tensor * a);
// cumulative sums along first axis (ne0)
GGML_API struct ggml_tensor * ggml_cumsum(
struct ggml_context * ctx,
struct ggml_tensor * a);
GGML_API struct ggml_tensor * ggml_conv_1d_dw_tts(
struct ggml_context * ctx,
struct ggml_tensor * a, // convolution kernel
struct ggml_tensor * b, // data
int s0, // stride
int p0, // padding
int d0); // dilation
GGML_API struct ggml_tensor * ggml_conv_transpose_1d_tts(
struct ggml_context * ctx,
struct ggml_tensor * a, // convolution kernel
struct ggml_tensor * b, // data
int s0, // stride
int p0, // padding
int d0, // dilation
int op0, // output padding
int g0); // groups
GGML_API struct ggml_tensor * ggml_conv_1d_tts(
struct ggml_context * ctx,
struct ggml_tensor * a, // convolution kernel
struct ggml_tensor * b, // data
int s0, // stride
int p0, // padding
int d0); // dilation
GGML_API struct ggml_tensor * ggml_stft(
struct ggml_context * ctx,
struct ggml_tensor * a,
struct ggml_tensor * w, // the window
int filter_length,
int hop_length,
bool compute_abs_and_angle);
GGML_API struct ggml_tensor * ggml_istft(
struct ggml_context * ctx,
struct ggml_tensor * a, // magnitude + phase
struct ggml_tensor * w,
int filter_length,
int hop_length,
bool from_abs_and_angle);
GGML_API struct ggml_tensor * ggml_upscale_linear(
struct ggml_context * ctx,
struct ggml_tensor * a,
int scale_factor);
//end kcpp dirtypatch
#ifdef __cplusplus
}
#endif

View file

@ -1668,6 +1668,776 @@ static void ggml_compute_forward_mul_mat_id(
}
/////////////////////////////////
///// kcpp: dirtypatch for tts cpp ////////////
inline static void ggml_vec_reci_f32 (const int n, float * y, const float * x) { for (int i = 0; i < n; ++i) y[i] = 1.0f/x[i]; }
inline static void ggml_vec_round_f32 (const int n, float * y, const float * x) { for (int i = 0; i < n; ++i) y[i] = (float)((int) (x[i] + 0.5f)); }
inline static void ggml_vec_mod_f32(const int n, float * y, const float * x, const float mod_val) { for (int i = 0; i < n; ++i) y[i] = fmod(x[i], mod_val); }
void ggml_compute_forward_sin_f32_ffast_math(struct ggml_tensor * dst);
static void ggml_compute_forward_reciprocal_f32(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
if (params->ith != 0) {
return;
}
GGML_ASSERT(ggml_are_same_shape(src0, dst));
const int n = ggml_nrows(src0);
const int nc = src0->ne[0];
GGML_ASSERT( dst->nb[0] == sizeof(float));
GGML_ASSERT(src0->nb[0] == sizeof(float));
for (int i = 0; i < n; i++) {
ggml_vec_reci_f32(nc,
(float *) ((char *) dst->data + i*( dst->nb[1])),
(float *) ((char *) src0->data + i*(src0->nb[1])));
}
}
static void ggml_compute_forward_reciprocal(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
switch (src0->type) {
case GGML_TYPE_F32:
{
ggml_compute_forward_reciprocal_f32(params, dst);
} break;
default:
{
GGML_ABORT("fatal error");
}
}
}
static void ggml_compute_forward_cumsum_f32(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
const int ith = params->ith;
const int nth = params->nth;
GGML_TENSOR_UNARY_OP_LOCALS;
if (ith > ne1) {
return;
}
const int rpt = (ne1 + nth - 1)/nth;
// row range for this thread
const int ir0 = rpt * ith;
const int ir1 = MIN(ir0 + rpt, ne1);
if (ir0 > ne1) {
return;
}
for (int b = 0; b < ne2; b++) {
for (int i1 = ir0; i1 < ir1; i1++) {
float running = 0.0f;
float * tgt_data = (float *)((char *) src0->data + i1*nb01 + b*nb02);
float * dst_data = (float *)((char *) dst->data + i1*nb1 + b*nb2);
for (int ii = 0; ii < ne0; ii++) {
running += tgt_data[ii];
dst_data[ii] = running;
}
}
}
}
static void ggml_compute_forward_cumsum(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
switch (src0->type) {
case GGML_TYPE_F32:
{
ggml_compute_forward_cumsum_f32(params, dst);
} break;
default:
{
GGML_ABORT("fatal error");
}
}
}
static void ggml_compute_forward_mod_f32(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
const int ith = params->ith;
if (ith != 0) {
return;
}
const float mod_val = ((float *) dst->op_params)[0];
GGML_ASSERT(ggml_are_same_shape(src0, dst));
const int n = ggml_nrows(src0);
const int nc = src0->ne[0];
GGML_ASSERT( dst->nb[0] == sizeof(float));
GGML_ASSERT(src0->nb[0] == sizeof(float));
for (int i = 0; i < n; i++) {
ggml_vec_mod_f32(nc,
(float *) ((char *) dst->data + i*( dst->nb[1])),
(float *) ((char *) src0->data + i*(src0->nb[1])),
mod_val);
}
}
static void ggml_compute_forward_mod(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
switch (src0->type) {
case GGML_TYPE_F32:
{
ggml_compute_forward_mod_f32(params, dst);
} break;
default:
{
GGML_ABORT("fatal error");
}
}
}
static void ggml_compute_forward_round_f32(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
if (params->ith != 0) {
return;
}
GGML_ASSERT(ggml_are_same_shape(src0, dst));
const int n = ggml_nrows(src0);
const int nc = src0->ne[0];
GGML_ASSERT( dst->nb[0] == sizeof(float));
GGML_ASSERT(src0->nb[0] == sizeof(float));
for (int i = 0; i < n; i++) {
ggml_vec_round_f32(nc,
(float *) ((char *) dst->data + i*( dst->nb[1])),
(float *) ((char *) src0->data + i*(src0->nb[1])));
}
}
static void ggml_compute_forward_round(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
switch (src0->type) {
case GGML_TYPE_F32:
{
ggml_compute_forward_round_f32(params, dst);
} break;
default:
{
GGML_ABORT("fatal error");
}
}
}
static void simple_dft(
float * mdst,
float * phdst,
float * buffer,
size_t n_fft,
size_t step) {
float base_k = M_PI * -2.0f / (float) n_fft;
for (int i = 0; i < n_fft; i++) {
float tm = 0.0f;
float tph = 0.0f;
for (int ii = 0; ii < n_fft; ii++) {
float k = base_k * (float)ii * (float)i;
float m = cosf(k);
float expk = sinf(k);
tm += mdst[ii*step] * m - phdst[ii*step] * expk;
tph += mdst[ii*step] * expk + phdst[ii*step] * m;
}
buffer[i*2] = tm;
buffer[i*2+1] = tph;
}
// assign magnitude and phase values from the accumulated buffer;
for (int i = 0; i < n_fft; i++) {
mdst[i*step] = buffer[i*2];
phdst[i*step] = buffer[i*2+1]; //-1.0e-15 < buffer[i*2+1] < 1.0e-15 ? 0.0f : buffer[i*2+1];
}
}
static void radix2_fft(
float * mdst,
float * phdst,
float * buffer,
size_t n_fft,
size_t step) {
if (n_fft == 1) {
return;
} else if (n_fft % 2 != 0) {
// rather than using chirp padding just fall back to dft when we have a size that isn't factorable by 2.
simple_dft(mdst, phdst, buffer, n_fft, step);
return;
}
radix2_fft(mdst, phdst, buffer, (size_t)n_fft/2, step*2);
radix2_fft(
(float*)((char*) mdst + step * sizeof(float)),
(float*)((char*) phdst + step * sizeof(float)),
buffer,
(size_t)n_fft/2,
step*2);
float km = M_PI * -2.0f / (float) n_fft;
for (int i = 0; 2 * i < n_fft; i++) {
float k = km * (float) i;
float k1 = cosf(k);
float k2 = sinf(k);
float mp = mdst[i*2*step];
float php = phdst[i*2*step];
float mq = mdst[(i*2+1)*step] * k1 - k2 * phdst[(i*2+1)*step];
float phq = mdst[(i*2+1)*step] * k2 + k1 * phdst[(i*2+1)*step];
buffer[i + n_fft] = php + phq;
buffer[i] = mp + mq;
buffer[(i + (n_fft / 2)) + n_fft] = php - phq;
buffer[(i + (n_fft / 2))] = mp - mq;
}
for (int i = 0; i < n_fft; i++) {
mdst[i*step] = buffer[i];
phdst[i*step] = buffer[i+n_fft];
}
}
static void ggml_compute_forward_stft_f32(
const struct ggml_compute_params * params,
struct ggml_tensor * dst,
bool compute_abs_and_angle) {
const struct ggml_tensor * src0 = dst->src[0];
const struct ggml_tensor * src1 = dst->src[1];
const float * w = (float*) src1->data;
// view src0 and dst with these strides and data offset inbytes during set
// nb0 is implicitly element_size because src0 and dst are contiguous
size_t n_fft = ((int32_t *) dst->op_params)[0];
size_t hop = ((int32_t *) dst->op_params)[1];
const int half = n_fft / 2;
GGML_ASSERT(src0->type == GGML_TYPE_F32);
GGML_ASSERT(src1->type == GGML_TYPE_F32);
GGML_ASSERT( dst->type == GGML_TYPE_F32);
GGML_TENSOR_BINARY_OP_LOCALS;
GGML_ASSERT(n_fft == ne10); // only currently supporting stft while n_fft is equal to the window size
const int ith = params->ith;
const int nth = params->nth;
if (ith == 0) {
// need to zero dst since we are accumulating into it
memset(dst->data, 0.0f, ggml_nbytes(dst));
memset(params->wdata, 0.0f, (n_fft * 2 + CACHE_LINE_SIZE_F32)*nth);
}
ggml_barrier(params->threadpool);
// it is faster to just use the max possible memory for the buffer rather than calculating the largest uneven half of the window.
float * buffer = ((float *) params->wdata) + (n_fft * 2 + CACHE_LINE_SIZE_F32) * ith;
// hops per thread
const int hpt = (ne1 + nth - 1)/nth;
// row range for this thread
const int ir0 = hpt * ith;
const int ir1 = MIN(ir0 + hpt, ne1);
for (int b = 0; b < ne01; b++) {
for (int i1 = ir0; i1 < ir1; i1++) {
const int ch = i1*hop;
float * mdst_data = (float *)((char *) dst->data + i1*nb1 + b*nb2);
float * phdst_data = (float *)((char *) dst->data + i1*nb1 + nb3 + b*nb2);
float * tgt_data = (float *)((char *) src0->data + b*nb01);
// preinitialize the magnitude data with the window applied;
for (int i = 0; i < n_fft; i++) {
int ai = (ch - half + i); // for handling reflective padding to support center view;
if (ai < 0) {
mdst_data[i] = tgt_data[-1 * ai] * w[i];
} else if (ai >= ne00) {
mdst_data[i] = tgt_data[ne00 - (ai - ne00 + 1)] * w[i];
} else {
mdst_data[i] = tgt_data[ai] * w[i];
}
}
radix2_fft(mdst_data, phdst_data, buffer, n_fft, 1);
if (compute_abs_and_angle) {
for (int i = 0; i < n_fft; i++) {
float abs = sqrtf(mdst_data[i]*mdst_data[i] + phdst_data[i]*phdst_data[i]);
float agl = atan2f(phdst_data[i], mdst_data[i]);
mdst_data[i] = abs;
phdst_data[i] = agl;
}
}
}
}
}
static void ggml_compute_forward_stft(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
switch (src0->type) {
case GGML_TYPE_F32:
{
ggml_compute_forward_stft_f32(params, dst, false);
} break;
default:
{
GGML_ABORT("fatal error");
}
}
}
static void ggml_compute_forward_abs_angle_stft(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
switch (src0->type) {
case GGML_TYPE_F32:
{
ggml_compute_forward_stft_f32(params, dst, true);
} break;
default:
{
GGML_ABORT("fatal error");
}
}
}
static void radix2_ifft(
float * mdst,
float * phdst,
float * buffer,
float * tgt,
float * window,
size_t n_fft,
size_t step,
int min_length,
int max_length,
int index,
int offset) {
radix2_fft(mdst, phdst, buffer, n_fft, step);
for (int i = 0; i < n_fft; i++) {
int base_index = (n_fft - i) % n_fft;
float w = window[base_index];
int tgt_index = base_index - offset;
int location = index + tgt_index;
if (location < min_length || location >= max_length) {
continue; // ignore reflective padding
}
tgt[location] += mdst[i] / n_fft * w;
}
}
static void ggml_compute_forward_istft_f32(
const struct ggml_compute_params * params,
struct ggml_tensor * dst,
bool from_abs_and_angle) {
const struct ggml_tensor * src0 = dst->src[0];
const struct ggml_tensor * window = dst->src[1];
// view src0 and dst with these strides and data offset inbytes during set
// nb0 is implicitly element_size because src0 and dst are contiguous
size_t n_fft = ((int32_t *) dst->op_params)[0];
size_t hop = ((int32_t *) dst->op_params)[1];
const int half = n_fft / 2;
GGML_ASSERT(src0->type == GGML_TYPE_F32);
GGML_ASSERT( dst->type == GGML_TYPE_F32);
GGML_TENSOR_UNARY_OP_LOCALS;
const int ith = params->ith;
const int nth = params->nth;
if (ith == 0) {
// need to zero dst since we are accumulating into it
memset(dst->data, 0, ggml_nbytes(dst));
}
ggml_barrier(params->threadpool);
// it is faster to just use the max possible memory for the buffer rather than calculating the largest uneven half of the window.
float * buffer = ((float *) params->wdata) + (n_fft * 2 + CACHE_LINE_SIZE_F32) * ith;
// this buffer is used to rebuild the two sided fft from a one sided intput as well compute the complex values from the magnitude and phase.
float * phm_buffer = ((float *) params->wdata) + (n_fft * 2 + CACHE_LINE_SIZE_F32) * nth + (n_fft * 2 + CACHE_LINE_SIZE_F32) * ith;
// frames per thread
const int spt = (ne0 + nth - 1)/nth;
// each thread effectively owns a portion of the output series, but distinct frames can overlap over series so we need to recompute overlapping ifft values
// around the edges.
const int poa = half / hop - 1;
const int pob = half / hop + 1;
// row range for this thread
const int min_spt = ith * spt;
const int max_spt = MIN(min_spt+spt, ne0);
const int ir0 = ith == 0 ? 0 : (min_spt / hop) - poa;
const int ir1 = MIN((max_spt / hop) + pob, ne01);
bool onesided = ne00 == half + 1;
for (int b = 0; b < ne02; b++) {
for (int i1 = ir0; i1 < ir1; i1++) {
float * mdst_data = (float *)((char *) src0->data + i1*nb01 + b*nb02);
float * phdst_data = (float *)((char *) src0->data + i1*nb01 + nb03 + b*nb02);
float * tgt_data = (float *)((char *) dst->data + b*nb1);
if (from_abs_and_angle || onesided) {
for (int i = 0; i < n_fft; i++) {
int index = i;
float multiplier = 1.0f;
if (onesided && i >= half + 1) {
index = n_fft - i;
multiplier = -1.0f;
}
float k = phdst_data[index];
float m = mdst_data[index];
float ph = m;
if (from_abs_and_angle) {
m *= cosf(k);
ph *= multiplier * sinf(k);
}
phm_buffer[i] = m;
phm_buffer[i+n_fft] = ph;
}
radix2_ifft(phm_buffer, phm_buffer + n_fft, buffer, tgt_data, (float*) window->data, n_fft, 1, min_spt, max_spt, i1*hop, half);
} else {
radix2_ifft(mdst_data, phdst_data, buffer, tgt_data, (float*) window->data, n_fft, 1, min_spt, max_spt, i1*hop, half);
}
// it should be noted that we don't remove any window applied via stft as that is better performed by a tensor multiplication op
}
}
}
static void ggml_compute_forward_abs_angle_istft(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
switch (src0->type) {
case GGML_TYPE_F32:
{
ggml_compute_forward_istft_f32(params, dst, true);
} break;
default:
{
GGML_ABORT("fatal error");
}
}
}
static void ggml_compute_forward_istft(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
switch (src0->type) {
case GGML_TYPE_F32:
{
ggml_compute_forward_istft_f32(params, dst, false);
} break;
default:
{
GGML_ABORT("fatal error");
}
}
}
static void ggml_compute_forward_conv_transpose_1d_f16_f32_tts(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
const struct ggml_tensor * src1 = dst->src[1];
GGML_ASSERT(src0->type == GGML_TYPE_F16);
GGML_ASSERT(src1->type == GGML_TYPE_F32);
GGML_ASSERT( dst->type == GGML_TYPE_F32);
GGML_TENSOR_BINARY_OP_LOCALS
const int ith = params->ith;
const int nth = params->nth;
const int nk = ne00*ne01*ne02;
GGML_ASSERT(nb00 == sizeof(ggml_fp16_t));
GGML_ASSERT(nb10 == sizeof(float));
if (ith == 0) {
memset(params->wdata, 0, params->wsize);
// permute kernel data (src0) from (K x Cout x Cin) to (Cin x K x Cout)
{
ggml_fp16_t * const wdata = (ggml_fp16_t *) params->wdata + 0;
for (int64_t i02 = 0; i02 < ne02; i02++) {
for (int64_t i01 = 0; i01 < ne01; i01++) {
const ggml_fp16_t * const src = (ggml_fp16_t *)((char *) src0->data + i02*nb02 + i01*nb01);
ggml_fp16_t * dst_data = wdata + i01*ne00*ne02;
for (int64_t i00 = 0; i00 < ne00; i00++) {
dst_data[i00*ne02 + i02] = src[i00];
}
}
}
}
// permute source data (src1) from (L x Cin) to (Cin x L)
{
ggml_fp16_t * const wdata = (ggml_fp16_t *) params->wdata + nk;
ggml_fp16_t * dst_data = wdata;
for (int64_t i11 = 0; i11 < ne11; i11++) {
const float * const src = (float *)((char *) src1->data + i11*nb11);
for (int64_t i10 = 0; i10 < ne10; i10++) {
dst_data[i10*ne11 + i11] = GGML_FP32_TO_FP16(src[i10]);
}
}
}
// need to zero dst since we are accumulating into it
memset(dst->data, 0, ggml_nbytes(dst));
}
ggml_barrier(params->threadpool);
const int32_t s0 = ((const int32_t*)(dst->op_params))[0];
const int32_t p0 = ((const int32_t*)(dst->op_params))[1];
const int32_t g0 = ((const int32_t*)(dst->op_params))[3];
// total rows in dst
const int nr = ne1;
const int gne02 = ne02 / g0;
const int kernel_m = gne02 == 1 ? 1 : ne02*ne00; // strictly speaking ne02*ne00 is wrong for other groupings but the grouping argument should be rarely used.
// rows per thread
const int dr = (nr + nth - 1)/nth;
// row range for this thread
const int ir0 = dr*ith;
const int ir1 = MIN(ir0 + dr, nr);
ggml_fp16_t * const wdata = (ggml_fp16_t *) params->wdata + 0;
ggml_fp16_t * const wdata_src = wdata + nk;
for (int i1 = ir0; i1 < ir1; i1++) {
float * dst_data = (float *)((char *) dst->data + i1*nb1);
ggml_fp16_t * wdata_kernel = wdata + i1*kernel_m;
for (int i10 = 0; i10 < ne10; i10++) {
const int i1n = i10*ne11 + (int)(i1 / gne02);
for (int i00 = 0; i00 < ne00; i00++) {
float v = 0;
if ((i10 * s0 < p0 && i00 >= p0) || (i10 * s0 >= p0 && i10 * s0 + i00 - p0 < ne0)) {
ggml_vec_dot_f16(gne02, &v, 0,
(ggml_fp16_t *) wdata_src + i1n, 0,
(ggml_fp16_t *) wdata_kernel + i00*ne02, 0, 1);
dst_data[i10*s0 + i00] += v;
}
}
}
}
}
static void ggml_compute_forward_conv_transpose_1d_f32_tts(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
const struct ggml_tensor * src1 = dst->src[1];
GGML_ASSERT(src0->type == GGML_TYPE_F32);
GGML_ASSERT(src1->type == GGML_TYPE_F32);
GGML_ASSERT( dst->type == GGML_TYPE_F32);
GGML_TENSOR_BINARY_OP_LOCALS
const int ith = params->ith;
const int nth = params->nth;
const int nk = ne00*ne01*ne02;
GGML_ASSERT(nb00 == sizeof(float));
GGML_ASSERT(nb10 == sizeof(float));
if (ith == 0) {
memset(params->wdata, 0, params->wsize);
// prepare kernel data (src0) from (K x Cout x Cin) to (Cin x K x Cout)
{
float * const wdata = (float *) params->wdata + 0;
for (int64_t i02 = 0; i02 < ne02; i02++) {
for (int64_t i01 = 0; i01 < ne01; i01++) {
const float * const src = (float *)((char *) src0->data + i02*nb02 + i01*nb01);
float * dst_data = wdata + i01*ne00*ne02;
for (int64_t i00 = 0; i00 < ne00; i00++) {
dst_data[i00*ne02 + i02] = src[i00];
}
}
}
}
// prepare source data (src1)
{
float * const wdata = (float *) params->wdata + nk;
float * dst_data = wdata;
for (int64_t i11 = 0; i11 < ne11; i11++) {
const float * const src = (float *)((char *) src1->data + i11*nb11);
for (int64_t i10 = 0; i10 < ne10; i10++) {
dst_data[i10*ne11 + i11] = src[i10];
}
}
}
// need to zero dst since we are accumulating into it
memset(dst->data, 0, ggml_nbytes(dst));
}
ggml_barrier(params->threadpool);
const int32_t s0 = ((const int32_t*)(dst->op_params))[0];
const int32_t p0 = ((const int32_t*)(dst->op_params))[1];
const int32_t g0 = ((const int32_t*)(dst->op_params))[3];
// total rows in dst
const int nr = ne1;
const int gne02 = ne02 / g0;
const int kernel_m = gne02 == 1 ? 1 : ne02*ne00; // strictly speaking ne02*ne00 is wrong for other groupings but the grouping argument should be rarely used.
// rows per thread
const int dr = (nr + nth - 1)/nth;
// row range for this thread
const int ir0 = dr*ith;
const int ir1 = MIN(ir0 + dr, nr);
float * const wdata = (float *) params->wdata + 0;
float * const wdata_src = wdata + nk;
for (int i1 = ir0; i1 < ir1; i1++) {
float * dst_data = (float *)((char *) dst->data + i1*nb1);
float * wdata_kernel = wdata + i1*kernel_m;
for (int i10 = 0; i10 < ne10; i10++) {
const int i1n = i10*ne11 + (int)(i1 / gne02);
for (int i00 = 0; i00 < ne00; i00++) {
float v = 0.0f;
if ((i10 * s0 < p0 && i00 >= p0) || (i10 * s0 >= p0 && i10 * s0 + i00 - p0 < ne0)) {
ggml_vec_dot_f32(gne02, &v, 0,
wdata_src + i1n, 0,
wdata_kernel + i00 * ne02, 0, 1);
dst_data[i10 * s0 + i00 - p0] += v;
}
}
}
}
}
static void ggml_compute_forward_conv_transpose_1d_tts(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
switch (src0->type) {
case GGML_TYPE_F16:
{
ggml_compute_forward_conv_transpose_1d_f16_f32_tts(params, dst);
} break;
case GGML_TYPE_F32:
{
ggml_compute_forward_conv_transpose_1d_f32_tts(params, dst);
} break;
default:
{
GGML_ABORT("fatal error");
}
}
}
static void ggml_compute_forward_upscale_linear_f32(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
GGML_ASSERT(src0->type == GGML_TYPE_F32);
const int ith = params->ith;
const int nth = params->nth;
GGML_TENSOR_UNARY_OP_LOCALS
// rows per thread
const int dr = (ne0 + nth - 1)/nth;
// row range for this thread
const int ir0 = dr*ith;
const int ir1 = MIN(ir0 + dr, ne0);
// the scale factor computed by dividing the output size by the input size
// this method currently doesn't support downscale interpolation (i.e. a scale factor between 0.0 and 1.0)
const float sf0 = (float) ne0 / src0->ne[0];
// in the PyTorch implementation of linear interpolation when performing linear upscaling, each side is padded by the half the scale factor of
// the first and last values respectively. Below we calculate half the scale factor.
const float hsf0 = sf0 / 2.0f;
const int64_t sf = (int64_t) sf0;
const int64_t hsf = (int64_t) hsf0;
for (int64_t i3 = 0; i3 < ne3; i3++) {
for (int64_t i2 = 0; i2 < ne2; i2 += nth) {
for (int64_t i1 = 0; i1 < ne1; i1++) {
float * y = (float *)((char *)dst->data + i1*nb1 + i2*nb2 + i3*nb3);
for (int64_t i0 = ir0; i0 < ir1; i0++) {
if (i0 < hsf) {
y[i0] = ((float *)((char *) src0->data + i1*nb01 + i2*nb02 + i3*nb03))[0];
continue;
} else if (i0 >= ne0 - hsf) {
y[i0] = ((float *)((char *) src0->data + i1*nb01 + i2*nb02 + i3*nb03))[src0->ne[0]-1];
continue;
}
const int64_t i00 = (i0 - hsf0) / sf0;
const float * x = (float *)((char *) src0->data + i00*nb00 + i1*nb01 + i2*nb02 + i3*nb03);
float base = x[0];
float top = x[1];
float diff_adj = (top - base) / sf0;
float adj = ((i0 - hsf) % sf) * diff_adj + (diff_adj / 2.0f);
y[i0] = base + adj;
}
}
}
}
}
static void ggml_compute_forward_upscale_linear(
const struct ggml_compute_params * params,
struct ggml_tensor * dst) {
const struct ggml_tensor * src0 = dst->src[0];
switch (src0->type) {
case GGML_TYPE_F32:
{
ggml_compute_forward_upscale_linear_f32(params, dst);
} break;
default:
{
GGML_ABORT("fatal error");
}
}
}
////// end kcpp dirtypatch /////////////
static void ggml_compute_forward(struct ggml_compute_params * params, struct ggml_tensor * tensor) {
GGML_ASSERT(params);
@ -2049,6 +2819,47 @@ static void ggml_compute_forward(struct ggml_compute_params * params, struct ggm
{
GGML_ABORT("fatal error");
}
//KCPP: dirtypatch for TTS cpp
case GGML_OP_RECIPROCAL:
{
ggml_compute_forward_reciprocal(params, tensor);
} break;
case GGML_OP_ROUND:
{
ggml_compute_forward_round(params, tensor);
} break;
case GGML_OP_CUMSUM:
{
ggml_compute_forward_cumsum(params, tensor);
} break;
case GGML_OP_MOD:
{
ggml_compute_forward_mod(params, tensor);
} break;
case GGML_OP_STFT:
{
ggml_compute_forward_stft(params, tensor);
} break;
case GGML_OP_AA_STFT:
{
ggml_compute_forward_abs_angle_stft(params, tensor);
} break;
case GGML_OP_ISTFT:
{
ggml_compute_forward_istft(params, tensor);
} break;
case GGML_OP_AA_ISTFT:
{
ggml_compute_forward_abs_angle_istft(params, tensor);
} break;
case GGML_OP_UPSCALE_LINEAR:
{
ggml_compute_forward_upscale_linear(params, tensor);
} break;
case GGML_OP_CONV_TRANSPOSE_1D_TTS:
{
ggml_compute_forward_conv_transpose_1d_tts(params, tensor);
} break;
}
}
@ -2356,6 +3167,19 @@ static int ggml_get_n_tasks(struct ggml_tensor * node, int n_threads) {
{
GGML_ABORT("fatal error");
}
case GGML_OP_RECIPROCAL: //kcpp: dirtypatch tts cpp
case GGML_OP_MOD:
case GGML_OP_ROUND:
case GGML_OP_CUMSUM:
case GGML_OP_STFT:
case GGML_OP_AA_STFT:
case GGML_OP_ISTFT:
case GGML_OP_AA_ISTFT:
case GGML_OP_UPSCALE_LINEAR:
case GGML_OP_CONV_TRANSPOSE_1D_TTS:
{
n_tasks = n_threads;
} break;
default:
{
fprintf(stderr, "%s: op not implemented: ", __func__);
@ -2777,6 +3601,7 @@ struct ggml_cplan ggml_graph_plan(
cur = ggml_type_size(GGML_TYPE_F32) * node->ne[0] * n_tasks;
} break;
case GGML_OP_CONV_TRANSPOSE_1D:
case GGML_OP_CONV_TRANSPOSE_1D_TTS: //kcpp: dirtypatch for tts
{
GGML_ASSERT(node->src[0]->ne[3] == 1);
GGML_ASSERT(node->src[1]->ne[2] == 1);
@ -2851,6 +3676,18 @@ struct ggml_cplan ggml_graph_plan(
{
GGML_ABORT("fatal error");
}
case GGML_OP_STFT:
case GGML_OP_AA_STFT: //kcpp: dirtypatch for tts cpp
{
cur = ggml_type_size(node->type)*(n_threads + node->ne[0] * n_threads * 2);
} break;
case GGML_OP_ISTFT:
case GGML_OP_AA_ISTFT:
{
// In order to support one sided reflective padding and complex conversions we need two extra n_fft sized buffers
// to prepare magnitude and phase for inverted FFTs.
cur = ggml_type_size(node->type)*(n_tasks + node->ne[0] * n_tasks * 4);
} break;
default:
break;
}

View file

@ -7062,3 +7062,201 @@ bool ggml_threadpool_params_match(const struct ggml_threadpool_params * p0, cons
if (p0->strict_cpu != p1->strict_cpu ) return false;
return memcmp(p0->cpumask, p1->cpumask, GGML_MAX_N_THREADS) == 0;
}
/////////////////// kcpp: dirtypatch of tts cpp ops //////////////////////
static struct ggml_tensor * ggml_reciprocal_impl(
struct ggml_context * ctx,
struct ggml_tensor * a,
bool inplace) {
struct ggml_tensor * result = inplace ? ggml_view_tensor(ctx, a) : ggml_dup_tensor(ctx, a);
result->op = GGML_OP_RECIPROCAL;
result->src[0] = a;
return result;
}
struct ggml_tensor * ggml_reciprocal(
struct ggml_context * ctx,
struct ggml_tensor * a) {
return ggml_reciprocal_impl(ctx, a, false);
}
struct ggml_tensor * ggml_reciprocal_inplace(
struct ggml_context * ctx,
struct ggml_tensor * a) {
return ggml_reciprocal_impl(ctx, a, true);
}
static int64_t calculate_number_of_frames(int64_t length, size_t hop_length) {
return (int64_t)(length / (int64_t) hop_length) + 1;
}
struct ggml_tensor * ggml_stft(
struct ggml_context * ctx,
struct ggml_tensor * a,
struct ggml_tensor * w,
int filter_length,
int hop_length,
bool compute_abs_and_angle) {
const int64_t ne[4] = { filter_length, calculate_number_of_frames(a->ne[0], hop_length), a->ne[1], 2};
struct ggml_tensor * result = ggml_new_tensor(ctx, GGML_TYPE_F32, 4, ne);
result->op = compute_abs_and_angle ? GGML_OP_AA_STFT : GGML_OP_STFT;
result->src[0] = a;
result->src[1] = w;
int32_t params[] = { (int32_t) filter_length, (int32_t) hop_length };
ggml_set_op_params(result, params, sizeof(params));
return result;
}
static int64_t calculate_original_length(int64_t frames, size_t hop_length) {
return (frames - 1) * (int64_t) hop_length;
}
struct ggml_tensor * ggml_istft(
struct ggml_context * ctx,
struct ggml_tensor * a,
struct ggml_tensor * w,
int filter_length,
int hop_length,
bool from_abs_and_angle) {
const int64_t ne[4] = { calculate_original_length(a->ne[1], hop_length), a->ne[2], 1, 1};
struct ggml_tensor * result = ggml_new_tensor(ctx, GGML_TYPE_F32, 4, ne);
result->op = from_abs_and_angle ? GGML_OP_AA_ISTFT : GGML_OP_ISTFT;
result->src[0] = a;
result->src[1] = w;
int32_t params[] = { (int32_t) filter_length, (int32_t) hop_length };
ggml_set_op_params(result, params, sizeof(params));
return result;
}
static struct ggml_tensor * ggml_round_impl(
struct ggml_context * ctx,
struct ggml_tensor * a,
bool inplace) {
struct ggml_tensor * result = inplace ? ggml_view_tensor(ctx, a) : ggml_dup_tensor(ctx, a);
result->op = GGML_OP_ROUND;
result->src[0] = a;
return result;
}
struct ggml_tensor * ggml_round(
struct ggml_context * ctx,
struct ggml_tensor * a) {
return ggml_round_impl(ctx, a, false);
}
struct ggml_tensor * ggml_round_inplace(
struct ggml_context * ctx,
struct ggml_tensor * a) {
return ggml_round_impl(ctx, a, true);
}
static struct ggml_tensor * ggml_mod_impl(
struct ggml_context * ctx,
struct ggml_tensor * a,
float mod_val,
bool inplace) {
struct ggml_tensor * result = inplace ? ggml_view_tensor(ctx, a) : ggml_dup_tensor(ctx, a);
result->op = GGML_OP_MOD;
result->src[0] = a;
ggml_set_op_params_f32(result, 0, mod_val);
return result;
}
struct ggml_tensor * ggml_mod(
struct ggml_context * ctx,
struct ggml_tensor * a,
float mod_val) {
return ggml_mod_impl(ctx, a, mod_val, false);
}
struct ggml_tensor * ggml_mod_inplace(
struct ggml_context * ctx,
struct ggml_tensor * a,
float mod_val) {
return ggml_mod_impl(ctx, a, mod_val, true);
}
struct ggml_tensor * ggml_cumsum(
struct ggml_context * ctx,
struct ggml_tensor * a) {
struct ggml_tensor * result = ggml_dup_tensor(ctx, a);
result->op = GGML_OP_CUMSUM;
result->src[0] = a;
return result;
}
GGML_API struct ggml_tensor * ggml_conv_1d_dw_tts(
struct ggml_context * ctx,
struct ggml_tensor * a,
struct ggml_tensor * b,
int s0,
int p0,
int d0) {
enum ggml_type t = b->type;
if (a->type == GGML_TYPE_F16 || b->type == GGML_TYPE_F16) {
t = GGML_TYPE_F16;
}
struct ggml_tensor * new_a = ggml_reshape_4d(ctx, a, a->ne[0], 1, a->ne[1], a->ne[2]);
struct ggml_tensor * new_b = ggml_reshape_4d(ctx, b, b->ne[0], 1, b->ne[1], b->ne[2]);
struct ggml_tensor * im2col = ggml_im2col(ctx, new_a, new_b, s0, 0, p0, 0, d0, 0, false, GGML_TYPE_F32);
struct ggml_tensor * result = ggml_mul_mat(ctx, im2col, a);
result = ggml_reshape_3d(ctx, result, b->ne[0], b->ne[1], 1);
return result;
}
GGML_API struct ggml_tensor * ggml_conv_1d_tts(
struct ggml_context * ctx,
struct ggml_tensor * a,
struct ggml_tensor * b,
int s0,
int p0,
int d0) {
enum ggml_type t = b->type;
if (a->type == GGML_TYPE_F16 || b->type == GGML_TYPE_F16) {
t = GGML_TYPE_F16;
}
struct ggml_tensor * im2col = ggml_im2col(ctx, a, b, s0, 0, p0, 0, d0, 0, false, t); // [N, OL, IC * K]
struct ggml_tensor * result =
ggml_mul_mat(ctx,
ggml_reshape_2d(ctx, im2col, im2col->ne[0], (im2col->ne[2] * im2col->ne[1])), // [N, OL, IC * K] => [N*OL, IC * K]
ggml_reshape_2d(ctx, a, (a->ne[0] * a->ne[1]), a->ne[2])); // [OCIC, K] => [OC, IC * K]
result = ggml_reshape_3d(ctx, result, im2col->ne[1], a->ne[2], im2col->ne[2]); // [N, OC, OL]
return result;
}
static int64_t ggml_calc_conv_transpose_1d_output_size_tts(int64_t ins, int64_t ks, int s, int p, int d, int op) {
return (ins - 1) * s - 2 * p + d * (ks - 1) + op + 1;
}
GGML_API struct ggml_tensor * ggml_conv_transpose_1d_tts(
struct ggml_context * ctx,
struct ggml_tensor * a,
struct ggml_tensor * b,
int s0,
int p0,
int d0,
int op0,
int g0) {
GGML_ASSERT(ggml_is_matrix(b));
GGML_ASSERT(a->ne[2] == b->ne[1]);
GGML_ASSERT(a->ne[3] == 1);
GGML_ASSERT((p0 < s0 || p0 == 0) && s0 % p0 == 0);
GGML_ASSERT(d0 == 1);
GGML_ASSERT(b->ne[1] % g0 == 0);
const int64_t ne[4] = {
ggml_calc_conv_transpose_1d_output_size_tts(b->ne[0], a->ne[0], s0, p0, 1 /*d0*/, op0),
a->ne[1]*g0, b->ne[2], 1,
};
struct ggml_tensor * result = ggml_new_tensor(ctx, GGML_TYPE_F32, 4, ne);
int32_t params[] = { s0, p0, d0, g0 }; // we don't need to pass output padding down to the op as it is apparent from the shape of the output.
ggml_set_op_params(result, params, sizeof(params));
result->op = GGML_OP_CONV_TRANSPOSE_1D_TTS;
result->src[0] = a;
result->src[1] = b;
return result;
}
struct ggml_tensor * ggml_upscale_linear(
struct ggml_context * ctx,
struct ggml_tensor * a,
int scale_factor) {
struct ggml_tensor * result = ggml_new_tensor_4d(ctx, a->type, a->ne[0] * scale_factor, a->ne[1], a->ne[2], a->ne[3]);
result->op = GGML_OP_UPSCALE_LINEAR;
result->src[0] = a;
return result;
}
//// end kcpp dirtypatch ////