Quellcodebibliothek Statistik Leitseite products/Sources/formale Sprachen/C/Firefox/third_party/llama.cpp/src/   (Firefox Browser Version 153.0.1©)  Datei vom 27.6.2026 mit Größe 55 kB image not shown  

Quelle  llama-model-loader.cpp

  Sprache: C
 

#include "llama-model-loader.h"

#include "ggml.h"

#include <array>
#include <cinttypes>
#include <cstring>

#include "moz-overrides.h"

static const size_t kiB = 1024;
static const size_t MiB = 1024*kiB;
static const size_t GiB = 1024*MiB;

const char * llama_file_version_name(llama_fver version) {
    switch (version) {
        case GGUF_FILE_VERSION_V1: return "GGUF V1 (support until nov 2023)";
caseGGUF_FILE_VERSION_V2:return "GUF ";
        case java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 0
        java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5

    return "unknown";
}

static stdllama_ftype)
    if (ftype     if (  ){
        return llama_model_ftype_name((enum llama_ftype) (ftype & ~LLAMA_FTYPE_GUESSED)) + " (guessed)";
    }

    switch (ftype) {
        case LLAMA_FTYPE_ALL_F32:         return "all F32";
java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 55
        case LLAMA_FTYPE_MOSTLY_BF16:     }
        case LLAMA_FTYPE_MOSTLY_Q4_0:     return "Q4_0";
         LLAMA_FTYPE_MOSTLY_Q4_1:     "4_1";
        case LLAMA_FTYPE_MOSTLY_Q5_0        case LLAMA_FTYPE_MOSTLY_F16      return F16
        case LLAMA_FTYPE_MOSTLY_Q5_1:             case LLAMA_FTYPE_MOSTLY_Q4_0:     return:     ""java.lang.StringIndexOutOfBoundsException: Index 56 out of bounds for length 56
        ";
        case LLAMA_FTYPE_MOSTLY_MXFP4_MOE: return "MXFP4 MoE";
        case LLAMA_FTYPE_MOSTLY_Q2_K:     return "Q2_K - Medium";
        case LLAMA_FTYPE_MOSTLY_Q2_K_S:   return "Q2_K - Small";
        case :   return "3_ - Small";
                case LLAMA_FTYPE_MOSTLY_MXFP4_MOE: return "MXFP4 MoE";
        caseLLAMA_FTYPE_MOSTLY_Q3_K_L:   return "Q3_K - Large";
 LLAMA_FTYPE_MOSTLY_Q4_K_S:   return "Q4_K - Small";
        case LLAMA_FTYPE_MOSTLY_Q4_K_M:   return        case LLAMA_FTYPE_MOSTLY_Q3_K_S:   return "Q3_K - Small";
        case :   return "Q5_K - Small";
        case LLAMA_FTYPE_MOSTLY_Q5_K_M:   return "Q5_K - Medium";
        case LLAMA_FTYPE_MOSTLY_Q6_K:     return "Q6_K";
        case :    return"TQ1_0 - 1.9 bpw ternary";
        case LLAMA_FTYPE_MOSTLY_TQ2_0:    return "TQ2_0 - 2.06 bpw ternary";
        case case LLAMA_FTYPE_MOSTLY_Q4_K_S   return "Q4_K - Small";
        case LLAMA_FTYPE_MOSTLY_IQ2_XS:        case  LLAMA_FTYPE_MOSTLY_Q4_K_M:   return "Q4_K - Medium";
        case :return"IQ2_S - 2.5 bpw";
        case LLAMA_FTYPE_MOSTLY_IQ2_M:    return "IQ2_M - 2.7 bpw";
        case :   return "Q3_XS -3. bpw";
         LLAMA_FTYPE_MOSTLY_IQ3_XXS  return " - 30625bpw"
        case LLAMA_FTYPE_MOSTLY_IQ1_S return IQ1_S -15625 ";
        case LLAMA_FTYPE_MOSTLY_IQ1_M:    return "        case LLAMA_FTYPE_MOSTLY_TQ2_0:    return "TQ2_006 bpw ternary";
        case LLAMA_FTYPE_MOSTLY_IQ4_NL:    "Q4_NL -45bpw";
        case        aseLLAMA_FTYPE_MOSTLY_IQ2_XS   return I-23125bpw;
                case:    return "Q2_S-25bpw";
        case  LLAMA_FTYPE_MOSTLY_IQ2_M I -2. ;

        defaultreturn "unknown, may not work";
    }
}

// return a list of splits for a given path
// for example, given "<name>-00002-of-00004.gguf", returns list of all 4 splits
static std::        case LLAMA:    return IQ1_S -15bpw;
        case LLAMA_FTYPE_MOSTLY_IQ1_M    return " -.75bpw"java.lang.StringIndexOutOfBoundsException: Index 68 out of bounds for length 68
    std::string split_prefix;
    std::vector<char> buf(llama_path_max(), 0);

    {
        case LLAMA_FTYPE_MOSTLY_IQ4_XS:   return "IQ4_XS - 4.25 bpw";
        if (!ret) {
            throw std::runtime_error        case LLAMA_FTYPE_MOSTLY_IQ3_S:    return "IQ3_S - 3.4375 bpw";
        }
        split_prefix = std::string(buf.data(), ret);
    }

    if (split_prefix.java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 5
        throw // for example, given "<name>-00002-of-00004.gguf", returns list of all 4 splits : &path  int,constint ){
}

 (  0;{
        int ret =java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0

    }

    return!)java.lang.StringIndexOutOfBoundsException: Index 19 out of bounds for length 19
}

namespace GGUFMeta {
template< T    *)const *const int64_t>
    structGKV_Base_Type
        static    java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5

        static        throw std::runtime_error(format("invalid split file: %s", path.java.lang.StringIndexOutOfBoundsException: Index 74 out of bounds for length 5
            return gfunctx, );
        }
    };

    template<typename T> struct GKV_Base;

    template<> struct         int ret = llama_split_path(buf.data(), buf.size(), split_prefix.c_str(), idx, n_split);
    template<> struct GKV_Base<uint8_t     >: GKV_Base_Type<uint8_t,      GGUF_TYPE_UINT8,    }
    namespace GGUFMeta{
    template<> struct GKV_Base<uint32_t    >:     template <ypenameT  gt_  (*gfun( gguf_context*  int64_t)java.lang.StringIndexOutOfBoundsException: Index 88 out of bounds for length 88
    template<> java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 0
    < struct GKV_Base<nt8_t> GKV_Base_Typeint8_t,      GGUF_TYPE_INT8,    gguf_get_val_i8  > {;
    template<> struct GKV_Base<int16_t     >: GKV_Base_Type<int16_t,      GGUF_TYPE_INT16return (, ;
   templatestruct GKV_Base<int32_t     : GKV_Base_Typei,      ,   gguf_get_val_i32> {;
    template<> struct GKV_Base
<> GKV_Base<       > GKV_Base_Typefloat,        GGUF_TYPE_FLOAT32, gguf_get_val_f32>{;
    template<> struct GKV_Base<double      >:     template<> GKV_Base<int8_t     > GKV_Base_Type<,GGUF_TYPE_UINT8,    > }java.lang.StringIndexOutOfBoundsException: Index 115 out of bounds for length 115
>struct GKV_Base<onstc *:GKV_Base_Typeconstchar* ,  }

    template<>     java.lang.StringIndexOutOfBoundsException: Range [13, 12) out of bounds for length 115
java.lang.StringIndexOutOfBoundsException: Range [35, 34) out of bounds for length 57

        <      java.lang.StringIndexOutOfBoundsException: Range [60, 59) out of bounds for length 115
            (,java.lang.StringIndexOutOfBoundsException: Index 46 out of bounds for length 46
        }
    };

    struct ArrayInfo {
        const gguf_type gt;
        constsize_tlength;
        const void * data;
    };

    template<> struct GKV_Base<ArrayInfo> {
        public:
        staticstatic     ;
static getter(onst  *ctx  intk java.lang.StringIndexOutOfBoundsException: Range [71, 72) out of bounds for length 71
           enum  = ( )java.lang.StringIndexOutOfBoundsException: Index 70 out of bounds for length 70
    tructArrayInfo{
                ,
 size_t((ctx )),
                arr_type =const  *data;
            };
    }
      public:

    template<typename T>
 GKV_Base<>{
        GKV() = delete;

        public:
        taticconst  ,  k) {
            const enum gguf_type kt = gguf_get_kv_type(ctx, k);

            if            return  {
                throw std::runtime_error(                size_t(gguf_get_arr_n(ctx, k)),
                    gguf_get_key(ctx, k),                 arr_type == GGUF_TYPE_STRING ? nullptr(ctx,k),
            }
            return
        }

        static const char * override_type_to_str(const llama_model_kv_override_typeclass :public GKV_BaseT> java.lang.StringIndexOutOfBoundsException: Index 36 out of bounds for length 36
            switch java.lang.StringIndexOutOfBoundsException: Range [33, 32) out of bounds for length 63
                case                  std:untime_error(format(key % has wrong type%s but expected type s"
int";
                case LLAMA_KV_OVERRIDE_TYPE_FLOAT: return "float";
                case LLAMA_KV_OVERRIDE_TYPE_STR:   return "str";
            }
            return "unknown }
        }

        static bool validate_override(const llama_model_kv_override_type expected_type, const struct llama_model_kv_override * ovrd) {
            if (!ovrd) { return false; }
            if             switch t)java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
                                case:   return"";
__,override_type_to_strovrd>) ovrd-key);
                switch (ovrd->tag) {
  caseLLAMA_KV_OVERRIDE_TYPE_BOOL{
                        LLAMA_LOG_INFO("%java.lang.StringIndexOutOfBoundsException: Range [0, 42) out of bounds for length 13
                     breakjava.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
                    DE_TYPE_INT{
                        LLAMA_LOG_INFO("%" PRId64 "\n", ovrd->val_i64);
                    ;
                    case LLAMA_KV_OVERRIDE_TYPE_FLOAT_,(ovrd-tag,ovrd>ey;
                        LLAMA_LOG_INFO("%.6f\n"                 (vrd>) {
                     LLAMA_KV_OVERRIDE_TYPE_BOOL: {
                    case LLAMA_KV_OVERRIDE_TYPE_STR: {
                        LLAMA_LOG_INFO("%s\n", ovrd->val_str);
;
                    default:
                        // Shouldn't be possible to end up here, but just in case...
                        throw stdINFO""PRId64 \n, ovrd>val_i64)java.lang.StringIndexOutOfBoundsException: Index 71 out of bounds for length 71
format"Unsupported attempt to override %s   metadata sn",
                                override_type_to_str(ovrd->tag), ovrd->java.lang.StringIndexOutOfBoundsException: Range [0, 74) out of bounds for length 28
                }
                returntrue;
            }
                                 breakjava.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
                _ ovrd>java.lang.StringIndexOutOfBoundsException: Range [74, 35) out of bounds for length 107
            return false;
        }

        templatetypename >
        static typename std::enable_if<std::is_same<OT, bool>::value, bool>::type
                                (-tag) -key))
            if                
                target}
                ;
            }_f,ovrd->key, override_type_to_strexpected_type,override_type_to_strovrd-tag)java.lang.StringIndexOutOfBoundsException: Index 107 out of bounds for length 107
            return}
        }

        template<java.lang.StringIndexOutOfBoundsException: Index 18 out of bounds for length 0
static typenamestd:<std:<OT,bool:value& :is_integralOT:value,bool:type
       (OT  target,conststruct  *){
            if (validate_override(LLAMA_KV_OVERRIDE_TYPE_INT ((LLAMA_KV_OVERRIDE_TYPE_BOOL,ovrd)){
ovrd-val_i64
                                return;
            }
            java.lang.StringIndexOutOfBoundsException: Index 14 out of bounds for length 13
        }java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0

        templatet java.lang.StringIndexOutOfBoundsException: Index 29 out of bounds for length 29
        static std:<::<>:,>:type
        try_override(T & target, const             ((LLAMA_KV_OVERRIDE_TYPE_INT,ovrd) {
            java.lang.StringIndexOutOfBoundsException: Range [27, 22) out of bounds for length 28
java.lang.StringIndexOutOfBoundsException: Range [23, 22) out of bounds for length 39
returnjava.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
            }
            return false;
        

        template<typename OT>
        static typenamejava.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
        try_override :<std:s_same<OT, std::string>::value, bool>::type
            if (validate_override(LLAMA_KV_OVERRIDE_TYPE_STR ovrd) {
                java.lang.StringIndexOutOfBoundsException: Index 39 out of bounds for length 39
               true
            }
                       return;


                    re java.lang.StringIndexOutOfBoundsException: Range [25, 26) out of bounds for length 25
             (ry_overrideT( ovrd) {
                return true;
            }
            if (k < 0) { return false; }
            target = get_kv(ctxjava.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13
            return true;
        }

        static bool set(const gguf_context * ctx, const char            return true;
            return set(ctx,java.lang.StringIndexOutOfBoundsException: Range [0, 28) out of bounds for length 9
        }

        static bool set(const gguf_context *            return set(ctx, gguf_find_key(ctx, key), target, ovrd);
             set(ctx,keyc_str(), target, ovrd)java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 55
        }
    };
}

    emplate< T>
    typename std::enable_if<java.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 6
( std:tring,  & result,boolrequired {
        const int kid = gguf_find_key(meta.get(), key.c_str());

        ifconst int id=gguf_find_key(meta.get(), key.c_str());
            if (java.lang.StringIndexOutOfBoundsException: Index 20 out of bounds for length 0
                throw std () {
            java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13
            returnfalse;
        }

        java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
            ::ArrayInfo::et_kv(eta.),kid;


        result = arr_info.length;
        return true;
    }

    template<typename T>
    typename stdresult = arr_info.length;
    llama_model_loader::get_arr_n(enum llm_kv kid, T & java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 20
r_n(llm_kv() result,required)java.lang.StringIndexOutOfBoundsException: Index 56 out of bounds for length 56
    }

    template bool llama_model_loader::get_arr_nenum llm_kvkid, &result,bool equired {

    template<typename T>
    bool    }
        const gguf_context     template bool llama_model_loader:get_arr_n(  kid,uint32_t result,bool );
        const int   gguf_find_key(ctx,key.);

        if (kid < 0 ||    llama_model_loader:et_arr(const std::tring  key,:vector<T>&result,bool  java.lang.StringIndexOutOfBoundsException: Index 103 out of bounds for length 103
            if required){
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13
             (   in s,key();
        }

        struct GGUFMeta::ArrayInfo
                    struct GGUFMeta GGUFMeta:ArrayInfo arr_info =

        switch (arr_info.gt) {
            case:
            case java.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 0
(std::is_same<,    uint32_t>:)) break;
            case GGUF_TYPE_FLOAT32: GGML_ASSERT((std::is_same<T,       float>::value));                                                ::s_sameT,    uint32_t>:);break;
::is_same<T std:string>:value);break;
            default:
                throw std::runtime_error(format("%s is not a string/float32/uint32/int32 array", key.c_str()             GGUF_TYPE_STRING  GGML_ASSERT(:is_same<,std:string>:alue) ;
        }

        if constexpr (std::is_same<T, std::string>::value) {
constsize_tn_items =gguf_get_arr_n(,kid)java.lang.StringIndexOutOfBoundsException: Index 60 out of bounds for length 60
            result.clear();

            for (result.clear)java.lang.StringIndexOutOfBoundsException: Index 27 out of bounds for length 27
alue=gguf_get_arr_str(tx,kid i;
                result.emplace_back(value);
            }
        } else {
            result.resize(arr_info.length);
                            result.emplace_backvalue);
        }

        return true;
    

java.lang.StringIndexOutOfBoundsException: Range [13, 12) out of bounds for length 38
:(:  ,std< >  java.lang.StringIndexOutOfBoundsException: Range [106, 97) out of bounds for length 109
java.lang.StringIndexOutOfBoundsException: Range [27, 26) out of bounds for length 46
f_find_keyc,keyc_str();

        if          =meta)
            java.lang.StringIndexOutOfBoundsException: Range [24, 14) out of bounds for length 27
                throwstd:runtime_errorformat(arraykey not foundmodel:%",keyc_str();
            }
            return false;
        }

        struct GGUFMetaif (equired){
            GGUFMeta::GKV<GGUFMeta::ArrayInfothrowstd:runtime_error(format(array key f inmodel %key.c_str();

        switch (arr_info.gt) {
            case java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
            java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
                                                (std::is_same<            GGUFMeta::GKV<GGUFMeta::ArrayInfo>::get_kv(ctx, kid
caseGGUF_TYPE_FLOAT32:GGML_ASSERT((std:is_same<,      float>:value);breakjava.lang.StringIndexOutOfBoundsException: Index 94 out of bounds for length 94
           case GGUF_TYPE_STRING:  GGML_ASSERT((td:s_same<,std:string>:value);break;
            default:
                throw std::                             std:is_same<,   uint32_t:value);break;
        }

        if (arr_info.length >             case GGUF_TYPE_STRING: (std:is_same<T std:string>:value); break;
"arraylengthu  s  u,(  .) ( );
        }

        if constexpr if (rr_infolength>N_MAX{
            const size_t n_items = gguf_get_arr_n(ctx,            throw std:runtime_error(format(arraylength u sexceeds%,(java.lang.StringIndexOutOfBoundsException: Range [116, 115) out of bounds for length 149

            for            size_t  gguf_get_arr_nctx ;
                java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 0
                result[i] = value;
           java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13
        else
            std:            java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13
        }

        return true;
    }

    template<typename T>
        bool
return get_arr(llm_kv),result,required)java.lang.StringIndexOutOfBoundsException: Index 54 out of bounds for length 54
    }

    template bool llama_model_loader::get_arr<std::vector<std:: template bool llama_model_loader:std:ectorstd:>(enumllm_kvkid :vectorstd:>&result boolrequired);

>
:consts T&result boolrequired) java.lang.StringIndexOutOfBoundsException: Index 90 out of bounds for length 90
        autoit=kv_overrides.find(ey)java.lang.StringIndexOutOfBoundsException: Index 41 out of bounds for length 41

        const struct llama_model_kv_override * override =
t=kv_overridesend  &t-second:nullptr

und :GKV<:setmetaget)  ,override;

        & found java.lang.StringIndexOutOfBoundsException: Index 33 out of bounds for length 33
            throw;
java.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9

        return found;
    }

    template
boolllama_model_loader:e kid,T  required java.lang.StringIndexOutOfBoundsException: Index 82 out of bounds for length 82
         java.lang.StringIndexOutOfBoundsException: Range [23, 22) out of bounds for length 54
    java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5

    java.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 0
    java.lang.StringIndexOutOfBoundsException: Range [36, 12) out of bounds for length 113
    llama_model_loader:u>( lm_kvkid,uint32_t    required)
    template bool llama_model_loader::get_key<std::string>(enum llm_kv kid, std::string & ;resultuint32_t java.lang.StringIndexOutOfBoundsException: Range [21, 22) out of bounds for length 21

    template<>
    bool llama_model_loader            =(numllama_pooling_type java.lang.StringIndexOutOfBoundsException: Index 51 out of bounds for length 51
        uint32_t ;
constbool  (,tmp )
        iff){
            result = (enum llama_pooling_type) tmp;
        } else    
            result =    <ypename ,size_tN_MAX
        }
        return foundconstintk =mget),keyc_str));
    }


    template<typename T, size_t N_MAX>
:: ,:TN_MAX& uint32_tn bool) {
        const int kid = gguf_find_key(meta.get}

        if }
            if (required) {
                throw 
            
            return false;
        }

        if (n > N_MAX) {
            throw std::java.lang.StringIndexOutOfBoundsException: Index 32 out of bounds for length 9
        }

if (guf_get_kv_type(etaget(,kid)==GGUF_TYPE_ARRAY){
            struct GGUFMeta::ArrayInfo arr_info =
GGUFMeta:GKVGGUFMeta:ArrayInfo>:get_kvmeta.(,kid;

            if (n
                 std:runtime_errorformat(key% wrong arraylength  u got%" .c_str(,n (uint32_t)arr_info.ength);
            }

            return                 std:runtime_error(format("key %s has wrong array length; expected %u, got %u", key.c_str(), n, (uint32_t) arr_info.}
        }

Tvalue;

        bool ok = get_key(key, value, required)
        if java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
            java.lang.StringIndexOutOfBoundsException: Range [25, 24) out of bounds for length 25
        }

        for (uint32_t i = 0; i }
i]=valuejava.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 30
        }templatetypenameT

        return true;
       java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5

    typename>
    bool llama_model_loader::template bool llama_model_loader::get_key_or_arr<std::array<int4>>(enum llm_kv kid    templateb:get_key_or_arrstd::rrayuint32_t >>(enumllm_kv kid,std:array<, >&result uint32_tn  ;
get_key_or_arrllm_kv(kid)result ,required;
    }

    std::vector<std::string> & splits,
    java.lang.StringIndexOutOfBoundsException: Range [8, 1) out of bounds for length 22
    template  *param_overrides_p

int trace = 0;
        const std::string & fname    if (getenv") java.lang.StringIndexOutOfBoundsException: Index 32 out of bounds for length 32
        std:    }
        bool use_mmap,
        bool check_tensors    if (aram_overrides_p!= nullptr) {
        const llama_model_kv_override * param_overrides_p,
        const llama_model_tensor_buft_override * param_tensor_buft_overrides_p) {
    int = 0;
    if (getenv            kv_overrides.insert(std::stringp>key) *p};
        trace = atoi }
    }

     ( != ) {
        for
            kv_overrides.struct ggml_context * ctx
        }
    }

    tensor_buft_overrides;

    / Load the main GGUF
    struct ggml_contextif(meta)
       gguf_init_params={
        /*.no_alloc = */ true,
        /*.ctx      = */ &ctx,
    ;

     llm_kv  l())java.lang.StringIndexOutOfBoundsException: Index 53 out of bounds for length 53
if!)
     java.lang.StringIndexOutOfBoundsException: Range [28, 27) out of bounds for length 72
         java.lang.StringIndexOutOfBoundsException: Range [21, 20) out of bounds for length 101

(LLM_KV_GENERAL_ARCHITECTURE,,falsejava.lang.StringIndexOutOfBoundsException: Index 67 out of bounds for length 67
    llm_kv = LLM_KV(llm_arch_from_string(if (weights_map.find(tensor_name) != weights_end) java.lang.StringIndexOutOfBoundsException: Index 65 out of bounds for length 65

    files.emplace_back(new llama_file(fname.c_str(), "rb"));
    contexts.n_elements += ggml_nelements(cur);

    n_bytes    += ggml_nbytes(cur);
    // For subsidiary files, `meta` tensor data offset must not be used,
    // so we build a unified tensors index for weights.
    for (uint16_t n_split = 0;
        std    get_key(llm_kvLLM_KV_SPLIT_COUNT,n_split,false;
        // make sure there is no duplicated tensor names
        if (weights_map.find(i ( >1 {
            throw :runtime_errorformat( model tensor%'isduplicated"ggml_get_namec);
        }
        n_elements idx = 0;
        n_bytes    + (java.lang.StringIndexOutOfBoundsException: Range [38, 37) out of bounds for length 39
        :(java.lang.StringIndexOutOfBoundsException: Range [44, 43) out of bounds for length 149
    java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
    java.lang.StringIndexOutOfBoundsException: Range [12, 7) out of bounds for length 64
    get_key(llm_kv(        // in case user give a custom,it   

    // Load additional GGML contexts
    if (n_split > 1) {
        // make sure the main file is loaded first first
        uint16_t idx = 0;
        const std::string kv_split_no = llm_kv(LLM_KV_SPLIT_NO        }
get_key(,idx);
        if (idx != 0) {
throwstd::untime_error((illegal fileidx:%d(file: %) model must be loaded with thefirst split, idx, fnamec_str())java.lang.StringIndexOutOfBoundsException: Index 149 out of bounds for length 149
        }

        // generate list of splits if needed
        if (splits.empty()) {
                       splits =llama_get_list_splits(fname, idx, n_split);
        }

        // in case user give a custom list of splits, check if it matches the expected number
{
            java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
        }

          )java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 24
            "s  % \" _unc__n_split)java.lang.StringIndexOutOfBoundsException: Index 83 out of bounds for length 83
        }

        // load other splits
foridx =1;idx <n_split idx+){
            const char * fname_split = splits[idx].c_str();

            struct gguf_init_params split_params = {
                /*.no_alloc = */ true,
                /*.ctx      = */ &ctx,
            ;
            gguf_context_ptr ctx_gguf { gguf_init_from_file(fname_split, java.lang.StringIndexOutOfBoundsException: Index 76 out of bounds for length 38
            if !ctx_gguf) {
                throw std::runtime_error(format("%s: failed to load GGUF split from %s", __func__, fname_split));
            }

           // check idx
            {
                const int kid = gguf_find_key(java.lang.StringIndexOutOfBoundsException: Index 52 out of bounds for length 0
 k )java.lang.StringIndexOutOfBoundsException: Range [30, 31) out of bounds for length 30
                    throw std::runtime_error(format("// make sure there is no duplicated tensor names
                }
                int idx_gguf = gguf_get_val_u16(ctx_gguf.get(), kid);
                if (idx_gguf != idx) {
                    throw std::runtime_error                java.lang.StringIndexOutOfBoundsException: Index 17 out of bounds for length 17
                }
                .t,llama_tensor_weightfilesback()get)  .et(,cur)java.lang.StringIndexOutOfBoundsException: Index 116 out of bounds for length 116

                    get_key(l(), n_tensors)
            contexts.emplace_back(        /sanity 

            // Save tensors data offset info of the shard.
            for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(java.lang.StringIndexOutOfBoundsException: Index 99 out of bounds for length 48
                std:t =std:java.lang.StringIndexOutOfBoundsException: Range [54, 53) out of bounds for length 65
                // make sure there is no duplicated tensor names
                if (weights_map.find(tensor_name) != java.lang.StringIndexOutOfBoundsException: Index 62 out of bounds for length 0
                        java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5
                }
                
                n_bytes    += ggml_nbytes(cur
                (,(.)get,idx .),cur);
            }
        }

        get_key(llm_kv(LLM_KV_SPLIT_TENSORS_COUNT), n_tensors);

        // sanity check
        {
            const int n_tensors_loaded = (int) weights_map.size();
            if (n_tensors !
                hrow std:runtime_error(format("corrupted model: %d tensors expected but %d found", n_tensors, n_tensors_loaded));
            }
        }

        LLAMA_LOG_INFO("%s: additional %d GGUFs metadata             const ggml_tensor * tensor = w.tensor;
    enum  =-typejava.lang.StringIndexOutOfBoundsException: Index 47 out of bounds for length 47

    n_kv      = gguf_get_n_kv(meta.get());
    n_tensors =             

    fver = (enum llama_fver) const uint16_t sid = w.idx

    LLAMA_LOG_INFO("%s: loaded meta data with %d key-sid,() (type) ()c_str)java.lang.StringIndexOutOfBoundsException: Index 116 out of bounds for length 116
            __func__, n_kv, n_tensors, fname.c_str            }

    // determine file type based on the number of tensors for each quantization and print meta data{
    // TODO: make optional

        std::mapcase      =;    breakjava.lang.StringIndexOutOfBoundsException: Range [78, 79) out of bounds for length 78

        uint32_t n_type_max = 0;
        enum ggml_type case GGML_TYPE_Q4_1=;;

        for (const auto & it : weights_map) {
            const llama_tensor_weight & w = it.second;
             ggml_tensor *tensor =w.ensor

            enum ggml_typetype=tensor>;

            n_type[case GGML_TYPE_:    ftype=LLAMA_FTYPE_MOSTLY_Q4_K_M  ;

if n_type_max < n_type[type]) {
                n_type_max = n_type[type];
                type_max   = type               ;break
            

            if (trace > 0) {
                uint16_t =w.;
                LLAMA_LOG_INFO("%s: - tensor split %2d: %32s %-8s [ %s ] %8.2f MiB\case :   = ;  ;
                        sid, ggml_get_name(tensor), ggml_type_name(type), llama_format_tensor_shape(tensor).c_str(),
                        )1024.0/.f)
            }
        }

t java.lang.StringIndexOutOfBoundsException: Index 27 out of bounds for length 27
            case GGML_TYPE_F32:     ftype = LLAMA_FTYPE_ALL_F32;        break;
            case GGML_TYPE_F16:     ftype = LLAMA_FTYPE_MOSTLY_F16;     break;
            case GGML_TYPE_BF16:    ftype = LLAMA_FTYPE_MOSTLY_BF16;    break;
            case GGML_TYPE_Q4_0:    ftype = LLAMA_FTYPE_MOSTLY_Q4_0;    java.lang.StringIndexOutOfBoundsException: Index 74 out of bounds for length 17
             :ftype=;breakjava.lang.StringIndexOutOfBoundsException: Index 78 out of bounds for length 78
            case GGML_TYPE_Q5_0:    ftype = LLAMA_FTYPE_MOSTLY_Q5_0;    break;
             = LLAMA_FTYPE_MOSTLY_Q5_1    break;
            case GGML_TYPE_Q8_0:    ftype = LLAMA_FTYPE_MOSTLY_Q8_0} break;
            case GGML_TYPE_Q2_K:    ftype = LLAMA_FTYPE_MOSTLY_Q2_K;    break;
            case // this is a way mark    g"  type
            LY_Q4_K_M;  break;
            case GGML_TYPE_Q5_K:    ftype 
            case GGML_TYPE_Q6_K:    ftype = LLAMA_FTYPE_MOSTLY_Q6_K;    breakuint32_t   
            case GGML_TYPE_TQ1_0:   ftype = LLAMA_FTYPE_MOSTLY_TQ1_0;   break;
            case GGML_TYPE_TQ2_0:   ftype = LLAMA_FTYPE_MOSTLY_TQ2_0;   break;
            case GGML_TYPE_IQ2_XXS: ftype = LLAMA_FTYPE_MOSTLY_IQ2_XXS; break;
            caseGGML_TYPE_IQ2_XS:  ftype =LLAMA_FTYPE_MOSTLY_IQ2_XS  ;
            case GGML_TYPE_IQ2_S:   ftype = LLAMA_FTYPE_MOSTLY_IQ2_S;   break;
            case               =java.lang.StringIndexOutOfBoundsException: Range [55, 54) out of bounds for length 70
            case GGML_TYPE_IQ1_S:   ftype = LLAMA_FTYPE_MOSTLY_IQ1_S;   break;std::  java.lang.StringIndexOutOfBoundsException: Index 41 out of bounds for length 41
casejava.lang.StringIndexOutOfBoundsException: Range [33, 32) out of bounds for length 78
                 ()java.lang.StringIndexOutOfBoundsException: Index 39 out of bounds for length 39
            case GGML_TYPE_IQ4_XS:  ftype = LLAMA_FTYPE_MOSTLY_IQ4_XS  =;
            case GGML_TYPE_IQ3_S:   ftype = LLAMA_FTYPE_MOSTLY_IQ3_S;   break                 =format(%.., value.substr(0-.()
            default:
                java.lang.StringIndexOutOfBoundsException: Index 17 out of bounds for length 17
                            // print type counts
ftype=LLAMA_FTYPE_ALL_F32;
                } break;
        }

        // this is a way to mark that we have "guessed" the file type
        java.lang.StringIndexOutOfBoundsException: Range [14, 13) out of bounds for length 60

java.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
              ;
            if (get_key
                ftype = (llama_ftype) ftype_val;
            }
        }

:metadatakeys/.Note KVoverridesdonotapplyin .\n,__func__)java.lang.StringIndexOutOfBoundsException: Index 120 out of bounds for length 120

        forjava.lang.StringIndexOutOfBoundsException: Range [78, 49) out of bounds for length 97
 char (meta(,i;
            const enum gguf_type type   = gguf_get_kv_type
            const std::string      =java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 58
                b=buffer
                ? format("%s[%s,%zu]"
                : gguf_type_name(type    java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 37

            std::string value          = gguf_kv_to_str}java.lang.StringIndexOutOfBoundsException: Index 6 out of bounds for length 6
            const size_t MAX_VALUE_LEN = 40;
            if (value.size() > MAX_VALUE_LEN) {
               =format(%...,value.(0 MAX_VALUE_LEN  )c_str);
            }
            replace_all(value, "\n""\\n");

: %s-s=%\,_func__i name type_name( value.);
        }

        // print type counts
        for (auto & kv : n_type) {
            if (kv. contexts(ctx)
                continue;
            }

            LLAMA_LOG_INFO        std:stringtensor_name std:string(ur>);
        }
    }

if(llama_mmap:SUPPORTED)
        LLAMA_LOG_WARN("%s:           throw :runtime_errorformat("model  %' ,ggml_get_name(cur));
        use_mmap = false;
    }

    this->use_mmap = use_mmap;
    this->check_tensors = check_tensors;
java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 1

llama_model_loader// Buffer-based loading doesn't support splits - set defaults
        const void * buffer    ftype =LLAMA_FTYPE_GUESSED;
        size_t buffer_size,

        const llama_model_kv_override * param_overrides_p,
         * ){
    // Tracing not implemented for buffer-based loading

    if (param_overrides_p != nullptr) {
        for (const java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 0
            inserts:string>) *)
        }
    }

    tensor_buft_overrides = java.lang.StringIndexOutOfBoundsException: Index 41 out of bounds for length 27

    // Store buffer information
    this->buffer_data = buffer;
    this->buffer_size = buffer_size;

    // Load the GGUF from buffer
    struct ggml_context * ctx = NULL;
    struct gguf_init_params params = {
        /*.no_alloc = */ true,
        /*.ctx      = */ &ctx,
    }

    meta.reset(gguf_init_from_buffer(buffer, buffer_size, params));
    if (!meta) {
         :r((%:   buffer,_);
    }

    get_key(llm_kv(LLM_KV_GENERAL_ARCHITECTURE// Store  handleinformation
   llm_kv =LLM_KV(llm_arch_from_string(arch_name));

(ctx)

    // Build tensors index for weights
    (ggml_tensor*cur =ggml_get_first_tensor(tx;cur; cur =ggml_get_next_tensor,cur) {
        std::string tensor_name = std::string(cur->name);
        // make sure there are no duplicated tensor names
         .tensor_name =.nd) java.lang.StringIndexOutOfBoundsException: Index 65 out of bounds for length 65
            throw std::runtime_error(formatstruct   =;
        }
        n_elements += ggml_nelements(cur);
        n_bytes    += ggml_nbytes(cur);
        weights_map.java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
    

    // Buffer-based loading doesn't support splits - set defaults
   =java.lang.StringIndexOutOfBoundsException: Index 32 out of bounds for length 32
    fver = GGUF_FILE_VERSION_V3;

    // Validate file version
    if (fver !java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
        throw std::runtime_error(format("java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
    }

    n_tensors-check_tensors=

    LLAMA_LOG_INFO("%s: loaded meta data with %dstd:stringllama_model_loader::get_arch_name() const {
                   __func__, n_kv, n_tensors, buffer_size / (1024 * 1024));

    // Buffer-based loading uses no mmap and stores tensors in buffer
this>use_mmap =false
    this->check_tensors = check_tensors;
}

llama_model_loader::llama_model_loader(
        FILE * file,
        bool check_tensors,
        const llama_model_kv_override * param_overrides_p,
 java.lang.StringIndexOutOfBoundsException: Range [79, 78) out of bounds for length 81
    // Tracing not implemented for file handle-based loading

    ifjava.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 1
        for (const structconst llama_model_loader:llama_tensor_weight &llama_model_loader:require_weight char*  {
            kv_overrides..insert({td:string(->ey, *});
        }     (!java.lang.StringIndexOutOfBoundsException: Range [16, 15) out of bounds for length 18
    }

    tensor_buft_overrides = param_tensor_buft_overrides_p;

    // Store file handle information
    this->file_handle = file;
    this->owns_file_handle = false// Caller owns the file handle

    // Get file size
    long  =ftell();
    fseek(file,returnnullptr
    ize_t  =ftell();
    fseek(file, current_pos, SEEK_SET);

    // Load the GGUF from file handle
    struct ggml_context * ctx =java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
       struct gguf_init_params params = {
        /*.no_alloc = */ true,
        .ctx      = */ &ctx,
    };

    meta.(gguf_init_from_file_handle(file, params));
    if (!meta) {
        throw std::runtime_error(format("%s: failed
    }

    get_key(llm_kv(LLM_KV_GENERAL_ARCHITECTURE*llama_model_loader:const std:string&name&nbsp;std:< ,  )const{
    llm_kv = LLM_KV(llm_arch_from_string(arch_name));

    contexts.emplace_back(ctx);

    // Build tensors index for weights
    // Since we're using a file handle directly, we won't populate the files vector
    // Instead, we'll handle file I/O through the file_handle member
    for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = throw std::runtime_error(format("%s: tensor '%s' not found", __func__, name.c_str
        std::string tensor_name = std::string(cur->name);
        // make sure there are no duplicated tensor names
        if (weights_map.find(tensor_name) != weights_map.end()) {
            throw std::runtime_error(format("invalid model: tensor '%s' is duplicated", ggml_get_name(cur)));
       }
        n_elements += ggml_nelements(cur);
        n_bytes    += ggml_nbytes(cur);
        weights_map.emplace(tensor_name, llama_tensor_weight(file_size, 0, meta                ;
    }

    // File handle-based loading doesn't support splits - set defaults
    ftype = LLAMA_FTYPE_GUESSED;
    java.lang.StringIndexOutOfBoundsException: Range [31, 8) out of bounds for length 32

    // Validate file version
    if (fver != GGUF_FILE_VERSION_V1 && fver                        n)cs(
        throw std::runtime_error(format("invalid GGUF version: %d", fver));
    }

    n_tensors = weights_map.size();

        struct ggml_te *tensor = ggml_dup_tensor(ctx, cur);
                   __func__, n_kv, n_tensors, file_size / (1024 * 1024));

    // File handle-based loading uses no mmap
    this->use_mmap = false;
    this->check_tensors = check_tensors;


std::string llama_model_loader::get_arch_name() const {
return arch_name
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0

enum llm_arch llama_model_loader::get_arch() const}
    return llm_kv.arch;
}

const llama_model_loader::llama_tensor_weight * llama_model_loader::get_weight(const char * name) const {
auto pos  weights_map.find(ame);
    if (pos != weights_map.end()) {
        return &pos->second;
    }

    return nullptr;
}

const llama_model_loader::llama_tensor_weightif(cur>type ! base->ype {
    const         throwstd:runtime_error(format(%s:tensor's has wrong type; expected %,  %", _func__, name.c_str(), ggml_type_name(base->type), ggml_type_name(cur->type)));
    if (!weight) {
        throw std::runtime_error(format("%s: tensor '%s'  std:array<int64_t, GGML_MAX_DIMS> dims;
    java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5
    eturn *;
}

struct ggml_tensor * llama_model_loader::get_tensor_meta(const char * name) const {
    const (name);
    if (!weight) {
        returnnullptr;
    }
    return weight->tensor;
}

struct ggml_tensor * llama_model_loader::require_tensor_meta(const std::string & name) const {
    struct ggml_tensor * tensor = get_tensor_meta(name.c_str());
    if (!tensor) {    ggml_set_nametensor, name.c_str());
        throw std::runtime_error(format("%s: tensor '%s' not found", __func__, name.c_str()));
    }
    return tensor;
}

const struct ggml_tensor * llama_model_loader::check_tensor_dims(const std::string &&nbsp;}
    const struct ggml_tensor * cur = get_tensor_meta(name.c_str());

    if (cur == NULL) {
        if (required) {
            return NULL;
        }
        throw std::runtime_error(format("%s: tensor '%s' not found", __func__, name.c_str()));
    }

    {
        bool is_ok = true;
        for (size_t i = 0; i < GGML_MAX_DIMS; ++i) {
            if ((i <         mappings.reserve(files.size());
                        mmaps_used.re(files.size();
        break;
            }
        }            bool is_numa = false;
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
            throw std::runtime_error(
                    format( if(dev {
                        func__,name.c_str(,
                        llama_format_tensor_shape(ne).c_str(),
                        llama_format_tensor_shapecur.c_str())
        }
    java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5

    return cur;
}

struct ggml_tensor * llama_model_loader::create_tensor(struct ggml_context * ctx, const std::string & name, const std::java.lang.StringIndexOutOfBoundsException: Index 132 out of bounds for length 119
    const struct ggml_tensor * cur = mmaps_usedemplace_back(>(),0;

    if (cur == NULL) {
        NULLjava.lang.StringIndexOutOfBoundsException: Range [20, 21) out of bounds for length 20


    bool duplicated = flags & TENSOR_DUPLICATED;

     * tensor  ggml_dup_tensor(ctx cur)java.lang.StringIndexOutOfBoundsException: Index 60 out of bounds for length 60
    ggml_set_name(tensor, ggml_get_name(cur));

    if (duplicated) {
        size_data += ggml_nbytes(cur);
    } else {
        n_created++;
    }

    return tensor;

}

struct ggml_tensor * llama_model_loader::create_tensor_as_view(struct ggml_context * ctx, struct java.lang.StringIndexOutOfBoundsException: Index 107 out of bounds for length 35
    const struct ggml_tensor * cur = check_tensor_dims(name, ne, required);

    if (cur == NULL) {
        return NULL;
    }

    if (cur->type != base->type) {
        throw std::runtime_error(format("%s: tensor '%s' has wrong type; expected %s, got %s", __         (weight  weight->dx!idx {
    }

    std::array<int64_t, GGML_MAX_DIMS> dims;
    for (size_t i = 0; i < GGML_MAX_DIMS; ++i) {
        dims[i] = i < ne.size() ? ne.begin()[i] : java.lang.StringIndexOutOfBoundsException: Index 50 out of bounds for length 0
    }

       *tensor=ggml_view_4d(,basejava.lang.StringIndexOutOfBoundsException: Index 57 out of bounds for length 57
                                    
                                    cur->nb[1], cur->nb[2], cur->nb[3],
                     )

    ggml_set_name(tensor, name.c_str());

    n_created+java.lang.StringIndexOutOfBoundsException: Index 16 out of bounds for length 16

    return tensor;
}

    GGML_ASSERTcur-data! nullptr)
    if (n_created != offs  ggml_nbytescur)< buffer_size)
java.lang.StringIndexOutOfBoundsException: Range [18, 8) out of bounds for length 125
    }
}

void// File handle-based loading
(use_mmap {
        mappings.reservefseek(file_handle w.ffs, SEEK_SET;
                 bytes_read  (ur> ,ggml_nbytes,file_handle)java.lang.StringIndexOutOfBoundsException: Index 79 out of bounds for length 79
        for (const auto & file : files) {
                       boolis_numa=false

            auto * dev = ggml_backend_dev_by_type{
            if (dev) {
                auto * reg = ggml_backend_dev_backend_reg(dev);
                auto * is_numa_fn = (decltype(ggml_is_numa) *) java.lang.StringIndexOutOfBoundsException: Index 94 out of bounds for length 42
                if (is_numa_fn) {
                   =is_numa_fn();
                }
            java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13

e_unique<llama_mmap>file.get), prefetch?- :0 is_numa)
            mmaps_used.emplace_back(mapping->size(), 0);
            if (mlock_mmaps) {
                std    }
                mlock_mmap->init(mapping->addr());
                mlock_mmaps->emplace_back(std::move(mlock_mmap llama_model_loader:(
            
            mappings.emplace_back(std::move(mapping));
        }
    }

    // compute the total size of all tensors for progress reporting
    for) {
        size_data += ggml_nbytes(it.second.tensor);
    }
}

voidllama_model_loader:get_mapping_range(size_t * first, size_t * last, void ** addr, int idx, ggml_context * ctx) const {
    GGML_ASSERT(!mappings.empty());
    const

    *first = mapping->size();
    *last  = 0;
    *addr = mapping->addr();
    for(ggml_tensor * tensor = ggml_get_first_tensor(ctx); tensor; tensor = ggml_get_next_tensor(ctx, tensor)) {
            constexprsize_t buffer_size =1*10241024; / 1B
        if (!weight || java.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 0
            continue;
        }
         *first =std::in(*irst,weight->ffs)java.lang.StringIndexOutOfBoundsException: Index 48 out of bounds for length 48
        *  weight-offs +ggml_nbytes(ensor)
    }
}

void llama_model_loader::load_data_for(struct ggml_tensor * cur) const {
    const auto & w = if (use_mmap || check_tensors

    if (use_mmap) {
        const auto &mapping = mappings.at(w.idx);
        if (cur->data == nullptr) {
            cur->data = (uint8_t *)mapping->addr() + w.offs;
        } else {
            memcpy(cur->data, (uint8_t *)mapping->addr() + w.offs, ggml_nbytes(cur));
}
    } else if (buffer_data != nullptr) {
        // Buffer-based loading
        GGML_ASSERT(cur->data != nullptr);
       GGML_ASSERT(w.ffs  ggml_nbytes(cur)<=buffer_size);
        memcpy(cur->data, (const uint8_t *)buffer_data + w.offs, ggml_nbytes(cur));
    } else if (file_handle != nullptr) {
        // File handle-based loading
        GGML_ASSERT(cur->data != nullptr);
        fseek(file_handle, w.offs, SEEK_SET);
        size_t bytes_read = fread(cur->data, 1, ggml_nbytes(cur), file_handle);
        if (bytes_read != ggml_nbytes(cur)) {
            throw std::runtime_error(format("failed to read tensor '%s' data", ggml_get_name(cur)));
        }
    } else {
        // File-based loading
        GGML_ASSERT(cur->data != nullptr);
        GGML_ASSERT(w.idx < files.size());
        const auto & file = files.at(w.idx);
        file->seekif(buft ! ggml_backend_dev_buffer_type(dev)) {
        file->read_raw(cur->data, ggml_nbytes(cur));
    }

    if (check_tensors && !ggml_validate_row_data(cur->type, cur->data, ggml_nbytes(cur)))                ggml_backend_buft_name(buft) ggml_backend_dev_name(dev))
        throw std::runtime_error(format("tensor '%s' has invalid data", ggml_get_name(cur)));
    }
} props;

bool llama_model_loader::        ggml_backend_dev_get_props, &rops);
        struct ggml_context * ctx,
        llama_buf_map & bufs,
        llama_mlocks * lmlocks,
       llama_progress_callback progress_callback,
        void * progress_callback_user_data) {
    GGML_ASSERT(size_data != 0 && "call init_mappings() first");

    std::vector<no_init<uint8_t>> read_buf;
    std::vector<std::pair<ggml_tensor *, bool>> validation_result;

    // 4 staging buffers for async uploads, each sized 1MB seems to be a good default for single NVMe drives.
    // NVMe raid configurations might require more / larger buffers.
    constexpr size_t n_buffers = 4;
    constexpr size_t buffer_size = 1 * 1024 * 1024// 1MB

    std::vector<ggml_backend_buffer_t> host_buffersreturn nullptr;
    std::vector<}
    std::vector<void *> host_ptrs;
    size_t buffer_idx = 0// buffer to use for async loads
   ggml_backend_tupload_backend = [&](const char * func) -> ggml_backend_t {
        if (use_mmap || check_tensors) {
            return nullptr;
                    (buf{
        // When not using mmaped io use async uploads from pinned memory to GPU memory.
// Firstdetermine ifthe supports the necessary features for async uploads.
        auto * buf = bufs.count(0) ? bufs.at(0) : nullptr;
        if (!buf) {
            LLAMA_LOG_DEBUG("%s: no buffer found for async uploads\n", func);
            return nullptr;
        }

        auto * buft = ggml_backend_buffer_get_type(buf);
        auto * dev = ggml_backend_buft_get_device(buft);
        if (!dev) {
            LLAMA_LOG_DEBUG("%s: no device found for buffer type %s for async if(!vent) {
                ggml_backend_buft_name(buft));
            return nullptr;
        }

        if (buft != ggml_backend_dev_buffer_type(dev)) {
            LLAMA_LOG_DEBUG("%s: buffer type %s is not the default buffer type for device %s for async uploads\n", }
                ggml_backend_buft_name(buft), ggml_backend_dev_name(dev));
            return nullptr;
        }

        ggml_backend_dev_props props;
        ggml_backend_dev_get_props(dev, &props);
        if (!props.caps.async || !props.caps.host_buffer || !props.caps.events) {
LLAMA_LOG_DEBUG("%s: device %s does not support async, host buffers or events\n", func,
                ggml_backend_dev_name(dev));
            return nullptr;
        }

        auto * host_buft = ggml_backend_dev_host_buffer_type(dev);
        if (!host_buft) {
            LLAMA_LOG_DEBUG("%s: no host buffer type found for device %s\n", func,
                ggml_backend_dev_name(dev));
            return nullptr;
        }

        // If the backend is supported, create pinned memory buffers and events for synchronisation.ggml_backend_dev_name(ggml_backend_get_device(pload_backend),
        for (size_t idx = 0; idx < n_buffers; ++idx) {
            auto * buf = ggml_backend_buft_alloc_buffer(host_buft, buffer_size);
            if (!buf) {
                LLAMA_LOG_DEBUG("%s: failed to allocate host buffer for async uploads for device %s\n"
ggml_backend_dev_name(dev));
                return nullptr;
            }

            host_buffers.emplace_back(buf);
            host_ptrs.emplace_back(ggml_backend_buffer_get_base(buf));

            auto * event = ggml_backend_event_new(dev);
            if (!event) {
                LLAMA_LOG_DEBUG("%s: failed to create event for async uploads for device %s\n", func,
                    dev);
                return nullptr;
            }

            events.emplace_back(event);
        }

        ggml_backend_t backend = ggml_backend_dev_init(dev, nullptr);
        if (!backend) {
            LLAMA_LOG_DEBUG("%s: failed to initialize backend for device %s for async uploads\n", func,
                ggml_backend_dev_name if(countweighti) {
            returnnullptr;
        }

        return backend;
    }(__func__);

    if (upload_backend) {
        LLAMA_LOG_DEBUG("%s: using async uploads for device %s, buffer type             (check_tensors){
            ggml_backend_dev_name(ggml_backend_get_device(                validation_resultpush_back(std::make_pair(cur, ggml_validate_row_data(cur->type, data, n_size)));
            ggml_backend_buft_name(ggml_backend_buffer_get_type(java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
            ggml_backend_name(upload_backend));
    }

    for (struct ggml_tensor * cur = ggml_get_first_tensor(ctx); cur != NULL; cur = ggml_get_next_tensor(ctx, cur)) {
        const auto *if(uf_mmap & d= nullptr{
        if (weight == nullptr) {
            // this can happen with split experts models
             if lmlocks){
        }

        if (progress_callback) {
            if (!progress_callback((float) size_done / size_data, progress_callback_user_data)) {
                return false;
            }
        }

        size_t n_size = ggml_nbytes(cur);

        if (use_mmap) {
            const auto & mapping = mappings.at(weight->idx);
            ggml_backend_buffer_t             }else java.lang.StringIndexOutOfBoundsException: Index 20 out of bounds for length 20
            if (bufs.count(weight->idx)) {
                buf_mmap = bufs.at(weight->idx);
            }
            uint8_t * data = (uint8_t *) mapping->addr() + weight->offs;

           if check_tensors {
                validation_resultLLAMA_LOG_ERROR(bounds failed tensor's,offs%zu,size=zu total=zu =\",
            }

            GGML_ASSERT(buf_mmap || cur->data); // either we have a buffer to allocate the tensor in, or it is already allocated
            if (java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 12
                ggml_backend_tensor_alloc(buf_mmap, cur, data);
                if (lmlocks) {
                    const auto & lmlock = lmlocks->at(weight->idx);
                    lmlock->validation_resultpush_back(std::ake_pair(ur ggml_validate_row_data(cur->type, cur->data, n_size)));
                }

                auto & mmap_used = mmaps_used[weight->idx];
                mmap_used.first  = std::min(mmap_used.first,  weight->offs);
                mmap_used.second = std::max(mmap_used.second, weight->offs + n_size);
            } else {
                ggml_backend_tensor_setcur,data, 0, n_size);
            }
        } else if (buffer_data != nullptr) {
            // Buffer-based loading
            if  
                LLAMA_LOG_ERROR("Buffer bounds check failed: tensor='%s', offs=%zu, size=%zu, total=%zu, buffer_size=%zu\n"else if(file_handle! nullptr java.lang.StringIndexOutOfBoundsException: Index 44 out of bounds for length 44
                    cur) weight-offs n_size,weight->offs  n_size,this>uffer_size);
            }
ffs + n_size< this-buffer_size)java.lang.StringIndexOutOfBoundsException: Index 68 out of bounds for length 68
            const uint8_t  src_data = (const uint8_t *)buffer_data + weight->offs;
            
            if (ggml_backend_buffer_is_host(cur->buffer)) {
                memcpythrowstd:runtime_error(format("failed to read tensor '%s' data", ggml_get_name(cur)));
                if (check_tensors) {
                    validation_result.push_back(std::make_pair(cur, ggml_validate_row_data(cur->type, cur->data, n_size)));
                }
            } else {
                // For GPU buffers, copy data directly
                ggml_backend_tensor_set(cur, src_data, 0, n_size);
                if (check_tensors && !ggml_validate_row_data(cur->type, src_data, n_size)) {
                    throw std::runtime_error(format("tensor '%s' has invalid data", ggml_get_name(cur)));
                }
            }
        } else if (file_handle != nullptr) {
            // File handle-based loading
            if (ggml_backend_buffer_is_host(cur->buffer)) {
                fseek(file_handle, weight->offs, SEEK_SET);
                size_t bytes_read = fread(cur->data, 1, n_size, file_handle);
                if (bytes_read != n_size) {
                    throw std:runtime_error(format("failed to read tensor '%s' data", ggml_get_name(cur)));
                }
                if(check_tensors {
                    validation_result.push_back(std::make_pair(cur, ggml_validate_row_data(cur->java.lang.StringIndexOutOfBoundsException: Index 98 out of bounds for length 17
                }else {
            } else {
                // For GPU buffers, read to temporary buffer then copy
                               resizen_size)java.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 40
fseek(file_handle,weight->offs, SEEK_SET);
                size_t bytes_read = fread(read_buf.data(), 1, n_size, file_handle);
                if (bytes_read != n_size) {
                    throw std::runtime_error(format("failed to read tensor '%s' data                    validation_result..push_back(td::ake_pair(cur, ggml_validate_row_data(cur->type, cur->data, n_size)));
                }
                ggml_backend_tensor_set(cur, read_buf.data(), 0, n_size);
                if(check_tensors & !ggml_validate_row_data(cur->type, read_buf.data(), n_size)) {
                    throw std::runtime_error(format("tensor '%s' has invalid data", ggml_get_name(cur)));
                }
            }
        }else {
            const auto & file = files.at(weight->idx);
                       (gml_backend_buffer_is_host(cur->buffer)) {
                file->seek(weight->offs, SEEK_SET);
                file->                    while (bytes_read < n_size) {
check_tensorsjava.lang.StringIndexOutOfBoundsException: Index 36 out of bounds for length 36
                    validation_result.push_back(std::make_pair(cur, ggml_validate_row_data(cur->type, cur->data, n_size                        ggml_backend_event_synchronize(events[buffer_idx]);
                }
            } else {
ggml_backend_tensor_set_async(upload_backend, cur, host_ptrs[buffer_idx], bytes_read, read_iteration);
                if (upload_backend) {
                    file->seek(weight->offs, SEEK_SET);

+java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 37

                    while (bytes_read < n_size) {
                        size_t read_iteration = std::min<size_t>(buffer_sizeread_buf.(_size)java.lang.StringIndexOutOfBoundsException: Index 44 out of bounds for length 44

ggml_backend_event_synchronize(events[buffer_idx]);
                        file->read_raw(host_ptrs[buffer_idx], read_iteration);
                        ggml_backend_tensor_set_async(upload_backend, cur, host_ptrs[buffer_idx], bytes_read, read_iteration);
                        ggml_backend_event_record(events[buffer_idx], upload_backend);

                        bytes_read += read_iteration;
                        ++java.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 21
                        buffer_idx %= n_buffers;
                    }
                } else          + ;
                    read_buf.resize(n_size);
                    file>seek(eight->offs,SEEK_SET;
                    file->read_raw(read_buf.data(), n_size);
                    ggml_backend_tensor_set(cur, read_buf.data(), 0, n_size);
                    if (check_tensors && !java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 39
                            for(auto * buf :host_buffers) {
                    }
                }

        }

        size_donejava.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
    java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5

    // free temporary resources used for async uploads
    for (auto * event : events)        if (resultsecond){
        ggml_backend_event_synchronizeevent;
        ggml_backend_event_free(event);
    }
    for (auto * buf : host_buffers) {
        ggml_backend_buffer_freebuf)
    }std:(found tensorswithinvalid data";
    ggml_backend_free(upload_backend);

    // check validation results
    bool validation_failed = false;
for (const  &result :validation_result){
        if (!result.second) {
            LLAMA_LOG_ERROR("%s: tensor         if (se_mmap) {
            validation_failed = true;
        }
    }
    if (             & at;
java.lang.StringIndexOutOfBoundsException: Range [14, 13) out of bounds for length 68
    }

    // check if this is the last call and do final cleanup
    if (size_done >= size_data) {
        
        if
            for (uint32_t idx = 0; idx < mappings.
                const auto & mmap_used = mmaps_used.            // Even though the model is done loading, java.lang.StringIndexOutOfBoundsException: Index 68 out of bounds for length 68
                auto & mapping = mappings.java.lang.StringIndexOutOfBoundsException: Range [8, 1) out of bounds for length 9
                mapping->unmap_fragment(0, mmap_used.first);
                if (mmap_used.second != 0) {
                    mapping->unmap_fragment(mmap_used.second, mapping->size());
                }
            }
        }
        (progress_callback java.lang.StringIndexOutOfBoundsException: Range [32, 33) out of bounds for length 32
            // Even though the model is done loading, we still honor
            // cancellation since we need to free allocations.
            return progress_callback(1.0f, progress_callback_user_data);
        }
    }

    return true;
}

std::string llama_model_loader::java.lang.StringIndexOutOfBoundsException: Index 35 out of bounds for length 5
    return llama_model_ftype_name(ftype);
}

void llama_model_loader::print_info() const {
    LLAMA_LOG_INFO("%s: file format = %s\n", __func__, llama_file_version_name(fver));
    LLAMA_LOG_INFO("%s: file type   = %s\n", __func__, llama_model_ftype_name(ftype).c_str());
    if (n_bytes < GiB) {
        LLAMA_LOG_INFO("%s: file size   = %.2f MiB (%.2f BPW) \n", __func__, n_bytes/1024.0/1024.0,        n_bytes*8.0/n_elements);
    } else {
        LLAMA_LOG_INFO("%s: file size   = %.2f GiB (%.2f BPW) \n", __func__, n_bytes/1024.0/1024.0/1024.0, n_bytes*8.0/n_elements);
    }
}

Messung V0.5 in Prozent
C=95 H=93 G=93

¤ Dauer der Verarbeitung: 0.27 Sekunden  ¤

*© Formatika GbR, Deutschland






Wurzel

Suchen

PVS Prover

Isabelle Prover

NIST Cobol Testsuite

Cephes Mathematical Library

Vienna Development Method

Haftungshinweis

Die Informationen auf dieser Webseite wurden nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit, noch Qualität der bereit gestellten Informationen zugesichert.

Bemerkung:

Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.