@@ -854,6 +854,85 @@ GGML_CALL static void ggml_backend_cuda_split_buffer_init_tensor([[maybe_unused]
854854}
855855
856856GGML_CALL static void ggml_backend_cuda_split_buffer_set_tensor ([[maybe_unused]] ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size) {
857+ if (!tensor->extra && tensor->view_src && tensor->view_src ->extra ) {
858+ // OK, this is an ugly hack, but I don't really see a way to trick the machine into correctly
859+ // loading non-contiguous merged split tensors.
860+ auto view_src = tensor->view_src ;
861+ auto extra = (ggml_split_tensor_t *)view_src->extra ;
862+ void * extra_ptr;
863+ memcpy (&extra_ptr, view_src->op_params , sizeof (extra_ptr));
864+ if (extra_ptr) {
865+ std::string merged_name = view_src->name ;
866+ if (auto pos = merged_name.find (" ffn_gate_up_exps.weight" ); pos != std::string::npos) {
867+ std::string name = tensor->name ;
868+ auto pos_u = name.find (" ffn_up_exps.weight" );
869+ auto pos_g = name.find (" ffn_gate_exps.weight" );
870+ if (pos_u != std::string::npos || pos_g != std::string::npos) {
871+ GGML_ASSERT (extra->split_dim == 1 );
872+ auto & ranges = *(const std::vector<std::vector<std::pair<int ,int >>> *)extra_ptr;
873+ int ne = 0 ;
874+ for (int is = 0 ; is < int (ranges.size ()); ++is) {
875+ auto & r = ranges[is];
876+ GGML_ASSERT ((extra->splits [is] && !r.empty ()) || (!extra->splits [is] && r.empty ()));
877+ if (r.empty ()) continue ;
878+ GGML_ASSERT (r.size () == 2 );
879+ auto split = extra->splits [is];
880+ ggml_cuda_set_device (is);
881+ int ir = pos_g != std::string::npos ? 0 : 1 ;
882+ auto p = r[ir];
883+ size_t offset = 0 ;
884+ if (ir == 1 ) {
885+ p.first -= tensor->ne [1 ];
886+ GGML_ASSERT (p.first >= 0 );
887+ offset = split->ne [1 ]/2 * split->nb [1 ];
888+ }
889+ for (int i02 = 0 ; i02 < split->ne [2 ]; ++i02) {
890+ auto dst = (char *)split->data + i02*split->nb [2 ] + offset;
891+ auto src = (const char *)data + i02*tensor->nb [2 ] + ne*tensor->nb [1 ];
892+ CUDA_CHECK (cudaMemcpyAsync (dst, src, p.second *tensor->nb [1 ], cudaMemcpyHostToDevice, cudaStreamPerThread));
893+ }
894+ ne += p.second ;
895+ CUDA_CHECK (cudaStreamSynchronize (cudaStreamPerThread));
896+ }
897+ }
898+ return ;
899+ }
900+ if (auto pos = merged_name.find (" ffn_gate_up_exps.bias" ); pos != std::string::npos) {
901+ std::string name = tensor->name ;
902+ auto pos_u = name.find (" ffn_up_exps.bias" );
903+ auto pos_g = name.find (" ffn_gate_exps.bias" );
904+ if (pos_u != std::string::npos || pos_g != std::string::npos) {
905+ GGML_ASSERT (extra->split_dim == 0 );
906+ auto & ranges = *(const std::vector<std::vector<std::pair<int ,int >>> *)extra_ptr;
907+ int ne = 0 ;
908+ for (int is = 0 ; is < int (ranges.size ()); ++is) {
909+ auto & r = ranges[is];
910+ GGML_ASSERT ((extra->splits [is] && !r.empty ()) || (!extra->splits [is] && r.empty ()));
911+ if (r.empty ()) continue ;
912+ GGML_ASSERT (r.size () == 2 );
913+ auto split = extra->splits [is];
914+ ggml_cuda_set_device (is);
915+ int ir = pos_g != std::string::npos ? 0 : 1 ;
916+ auto p = r[ir];
917+ size_t offset = 0 ;
918+ if (ir == 1 ) {
919+ p.first -= tensor->ne [0 ];
920+ GGML_ASSERT (p.first >= 0 );
921+ offset = split->ne [0 ]/2 * split->nb [0 ];
922+ }
923+ for (int i01 = 0 ; i01 < split->ne [1 ]; ++i01) {
924+ auto dst = (char *)split->data + i01*split->nb [1 ] + offset;
925+ auto src = (const char *)data + i01*tensor->nb [1 ] + ne*tensor->nb [0 ];
926+ CUDA_CHECK (cudaMemcpyAsync (dst, src, p.second *tensor->nb [0 ], cudaMemcpyHostToDevice, cudaStreamPerThread));
927+ }
928+ ne += p.second ;
929+ CUDA_CHECK (cudaStreamSynchronize (cudaStreamPerThread));
930+ }
931+ }
932+ return ;
933+ }
934+ }
935+ }
857936 if (!tensor->extra ) return ;
858937 static std::map<ggml_type, int > k_map = {
859938 { GGML_TYPE_Q4_0_R8 , 8 },
@@ -886,7 +965,6 @@ GGML_CALL static void ggml_backend_cuda_split_buffer_set_tensor([[maybe_unused]]
886965 { GGML_TYPE_Q8_KV_R8 , 4 },
887966 { GGML_TYPE_Q8_K_R8 , 8 },
888967 };
889- // printf("%s(%s)\n", __func__, tensor->name);
890968
891969 // split tensors must always be set in their entirety at once
892970 GGML_ASSERT (offset == 0 );
@@ -984,7 +1062,6 @@ GGML_CALL static void ggml_backend_cuda_split_buffer_set_tensor([[maybe_unused]]
9841062 auto row_size = ggml_row_size (tensor->type , tensor->ne [0 ]);
9851063 std::vector<char > host_buffer;
9861064 int ne1 = 0 ;
987- int extra_ne1 = 0 ;
9881065 for (int i = 0 ; i < extra->n_device ; ++i) {
9891066 auto split = extra->splits [i];
9901067 if (!split) continue ;
0 commit comments