1
0

op_rope_f16.comp 2.8 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273
  1. #version 450
  2. #include "rope_common.comp"
  3. layout(binding = 0) buffer restrict readonly tensorInA { float16_t inA[]; };
  4. layout(binding = 1) buffer restrict readonly tensorInB { int inB[]; };
  5. layout(binding = 2) buffer restrict writeonly tensorOut { float16_t out_[]; };
  6. void main() {
  7. const uint i3 = gl_WorkGroupID.z;
  8. const uint i2 = gl_WorkGroupID.y;
  9. const uint i1 = gl_WorkGroupID.x;
  10. const bool is_neox = (pcs.mode & 2) != 0;
  11. float corr_dims[2];
  12. rope_yarn_corr_dims(pcs.n_dims, pcs.n_orig_ctx, pcs.freq_base, pcs.beta_fast, pcs.beta_slow, corr_dims);
  13. const float theta_scale = pow(pcs.freq_base, -2.0/pcs.n_dims);
  14. const int p = inB[pcs.inBOff + i2];
  15. float theta = float(p);
  16. if (!is_neox) {
  17. for (uint i0 = 0; i0 < pcs.ne0; i0 += 2) {
  18. float cos_theta, sin_theta;
  19. rope_yarn(theta, pcs.freq_scale, corr_dims, i0, pcs.ext_factor, pcs.attn_factor, cos_theta, sin_theta);
  20. theta *= theta_scale;
  21. const uint src = uint((i3*pcs.nb03 + i2*pcs.nb02 + i1*pcs.nb01 + i0*pcs.nb00) / 2) + pcs.inAOff; // Based from in
  22. const uint dst_data = uint((i3*pcs.nb3 + i2*pcs.nb2 + i1*pcs.nb1 + i0*pcs.nb0) / 2) + pcs.outOff; // Based from out_
  23. const float x0 = float(inA[src]);
  24. const float x1 = float(inA[src+1]);
  25. out_[dst_data] = float16_t(x0*cos_theta - x1*sin_theta);
  26. out_[dst_data+1] = float16_t(x0*sin_theta + x1*cos_theta);
  27. }
  28. } else {
  29. const float inv_ndims = -1.f/pcs.n_dims;
  30. for (uint ic = 0; ic < pcs.n_dims; ic += 2) {
  31. const uint cur_rot = ic;
  32. float cos_theta, sin_theta;
  33. rope_yarn(theta, pcs.freq_scale, corr_dims, cur_rot, pcs.ext_factor, pcs.attn_factor, cos_theta, sin_theta);
  34. theta *= theta_scale;
  35. const uint i0 = ic/2;
  36. const uint src = uint((i3*pcs.nb03 + i2*pcs.nb02 + i1*pcs.nb01 + i0*pcs.nb00) / 2) + pcs.inAOff; // Based from in
  37. const uint dst_data = uint((i3*pcs.nb3 + i2*pcs.nb2 + i1*pcs.nb1 + i0*pcs.nb0) / 2) + pcs.outOff; // Based from out_
  38. const float x0 = float(inA[src]);
  39. const float x1 = float(inA[src+pcs.n_dims/2]);
  40. out_[dst_data] = float16_t(x0*cos_theta - x1*sin_theta);
  41. out_[dst_data+pcs.n_dims/2] = float16_t(x0*sin_theta + x1*cos_theta);
  42. }
  43. for (uint ic = pcs.n_dims; ic < pcs.ne0; ic += 2) {
  44. const uint i0 = ic;
  45. const uint src = uint((i3*pcs.nb03 + i2*pcs.nb02 + i1*pcs.nb01 + i0*pcs.nb00) / 2) + pcs.inAOff; // Based from in
  46. const uint dst_data = uint((i3*pcs.nb3 + i2*pcs.nb2 + i1*pcs.nb1 + i0*pcs.nb0) / 2) + pcs.outOff; // Based from out_
  47. out_[dst_data + 0] = inA[src + 0];
  48. out_[dst_data + 1] = inA[src + 1];
  49. }
  50. }
  51. }