constchar * llama_file_version_name(llama_fver version) { switch (version) { case GGUF_FILE_VERSION_V1: return"GGUF V1 (support until nov 2023)"; caseGGUF_FILE_VERSION_V2:return"GUF "; case java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 0
java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5
return"unknown";
}
static stdllama_ftype) if (ftype if ( ){ return llama_model_ftype_name((enum llama_ftype) (ftype & ~LLAMA_FTYPE_GUESSED)) + " (guessed)";
}
switch (ftype) { case LLAMA_FTYPE_ALL_F32: return"all F32";
java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 55 case LLAMA_FTYPE_MOSTLY_BF16: } case LLAMA_FTYPE_MOSTLY_Q4_0: return"Q4_0";
LLAMA_FTYPE_MOSTLY_Q4_1: "4_1"; case LLAMA_FTYPE_MOSTLY_Q5_0 case LLAMA_FTYPE_MOSTLY_F16 return F16 case LLAMA_FTYPE_MOSTLY_Q5_1: case LLAMA_FTYPE_MOSTLY_Q4_0: return: ""java.lang.StringIndexOutOfBoundsException: Index 56 out of bounds for length 56 "; case LLAMA_FTYPE_MOSTLY_MXFP4_MOE: return"MXFP4 MoE"; case LLAMA_FTYPE_MOSTLY_Q2_K: return"Q2_K - Medium"; case LLAMA_FTYPE_MOSTLY_Q2_K_S: return"Q2_K - Small"; case : return"3_ - Small"; case LLAMA_FTYPE_MOSTLY_MXFP4_MOE: return"MXFP4 MoE"; caseLLAMA_FTYPE_MOSTLY_Q3_K_L: return"Q3_K - Large";
LLAMA_FTYPE_MOSTLY_Q4_K_S: return"Q4_K - Small"; case LLAMA_FTYPE_MOSTLY_Q4_K_M: returncase LLAMA_FTYPE_MOSTLY_Q3_K_S: return"Q3_K - Small"; case : return"Q5_K - Small"; case LLAMA_FTYPE_MOSTLY_Q5_K_M: return"Q5_K - Medium"; case LLAMA_FTYPE_MOSTLY_Q6_K: return"Q6_K"; case : return"TQ1_0 - 1.9 bpw ternary"; case LLAMA_FTYPE_MOSTLY_TQ2_0: return"TQ2_0 - 2.06 bpw ternary"; case case LLAMA_FTYPE_MOSTLY_Q4_K_S return"Q4_K - Small"; case LLAMA_FTYPE_MOSTLY_IQ2_XS: case LLAMA_FTYPE_MOSTLY_Q4_K_M: return"Q4_K - Medium"; case :return"IQ2_S - 2.5 bpw"; case LLAMA_FTYPE_MOSTLY_IQ2_M: return"IQ2_M - 2.7 bpw"; case : return"Q3_XS -3. bpw";
LLAMA_FTYPE_MOSTLY_IQ3_XXS return" - 30625bpw" case LLAMA_FTYPE_MOSTLY_IQ1_S return IQ1_S -15625"; case LLAMA_FTYPE_MOSTLY_IQ1_M: return" case LLAMA_FTYPE_MOSTLY_TQ2_0: return "TQ2_006 bpw ternary"; case LLAMA_FTYPE_MOSTLY_IQ4_NL: "Q4_NL -45bpw"; case aseLLAMA_FTYPE_MOSTLY_IQ2_XS return I-23125bpw;
case: return"Q2_S-25bpw"; case LLAMA_FTYPE_MOSTLY_IQ2_M I -2. ;
default: return"unknown, may not work";
}
}
// return a list of splits for a given path // for example, given "<name>-00002-of-00004.gguf", returns list of all 4 splits static std:: case LLAMA: return IQ1_S -15bpw; case LLAMA_FTYPE_MOSTLY_IQ1_M return" -.75bpw"java.lang.StringIndexOutOfBoundsException: Index 68 out of bounds for length 68
std::string split_prefix;
std::vector<char> buf(llama_path_max(), 0);
{ case LLAMA_FTYPE_MOSTLY_IQ4_XS: return"IQ4_XS - 4.25 bpw"; if (!ret) { throw std::runtime_error case LLAMA_FTYPE_MOSTLY_IQ3_S: return"IQ3_S - 3.4375 bpw";
}
split_prefix = std::string(buf.data(), ret);
}
if (split_prefix.java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 5 throw// for example, given "<name>-00002-of-00004.gguf", returns list of all 4 splits : &path int,constint ){
}
( 0;{ int ret =java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
}
return!)java.lang.StringIndexOutOfBoundsException: Index 19 out of bounds for length 19
}
namespace GGUFMeta { template< T *)const *const int64_t> structGKV_Base_Type static java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5
staticthrow std::runtime_error(format("invalid split file: %s", path.java.lang.StringIndexOutOfBoundsException: Index 74 out of bounds for length 5 return gfunctx, );
}
};
template<typename T> struct GKV_Base;
template<> struct int ret = llama_split_path(buf.data(), buf.size(), split_prefix.c_str(), idx, n_split); template<> struct GKV_Base<uint8_t >: GKV_Base_Type<uint8_t, GGUF_TYPE_UINT8, } namespace GGUFMeta{ template<> struct GKV_Base<uint32_t >: template <ypenameT gt_ (*gfun( gguf_context* int64_t)java.lang.StringIndexOutOfBoundsException: Index 88 out of bounds for length 88 template<> java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 0
< struct GKV_Base<nt8_t> GKV_Base_Typeint8_t, GGUF_TYPE_INT8, gguf_get_val_i8 > {; template<> struct GKV_Base<int16_t >: GKV_Base_Type<int16_t, GGUF_TYPE_INT16return (, ; template< struct GKV_Base<int32_t : GKV_Base_Typei, , gguf_get_val_i32> {; template<> struct GKV_Base
<> GKV_Base< > GKV_Base_Typefloat, GGUF_TYPE_FLOAT32, gguf_get_val_f32>{; template<> struct GKV_Base<double >: template<> GKV_Base<int8_t > GKV_Base_Type<,GGUF_TYPE_UINT8, > }java.lang.StringIndexOutOfBoundsException: Index 115 out of bounds for length 115
>struct GKV_Base<onstc *:GKV_Base_Typeconstchar* , }
template<> java.lang.StringIndexOutOfBoundsException: Range [13, 12) out of bounds for length 115
java.lang.StringIndexOutOfBoundsException: Range [35, 34) out of bounds for length 57
< java.lang.StringIndexOutOfBoundsException: Range [60, 59) out of bounds for length 115
(,java.lang.StringIndexOutOfBoundsException: Index 46 out of bounds for length 46
}
};
staticconstchar * override_type_to_str(const llama_model_kv_override_typeclass :public GKV_BaseT> java.lang.StringIndexOutOfBoundsException: Index 36 out of bounds for length 36 switch java.lang.StringIndexOutOfBoundsException: Range [33, 32) out of bounds for length 63 case std:untime_error(format(key % has wrong type%s but expected type s" int"; case LLAMA_KV_OVERRIDE_TYPE_FLOAT: return"float"; case LLAMA_KV_OVERRIDE_TYPE_STR: return"str";
} return"unknown }
}
staticbool validate_override(const llama_model_kv_override_type expected_type, conststruct llama_model_kv_override * ovrd) { if (!ovrd) { returnfalse; } ifswitch t)java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
case: return"";
__,override_type_to_strovrd>) ovrd-key); switch (ovrd->tag) { caseLLAMA_KV_OVERRIDE_TYPE_BOOL{
LLAMA_LOG_INFO("%java.lang.StringIndexOutOfBoundsException: Range [0, 42) out of bounds for length 13 breakjava.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
DE_TYPE_INT{
LLAMA_LOG_INFO("%" PRId64 "\n", ovrd->val_i64);
; case LLAMA_KV_OVERRIDE_TYPE_FLOAT_,(ovrd-tag,ovrd>ey;
LLAMA_LOG_INFO("%.6f\n" (vrd>) {
LLAMA_KV_OVERRIDE_TYPE_BOOL: { case LLAMA_KV_OVERRIDE_TYPE_STR: {
LLAMA_LOG_INFO("%s\n", ovrd->val_str);
; default: // Shouldn't be possible to end up here, but just in case... throw stdINFO""PRId64 \n, ovrd>val_i64)java.lang.StringIndexOutOfBoundsException: Index 71 out of bounds for length 71
format"Unsupported attempt to override %s metadata sn",
override_type_to_str(ovrd->tag), ovrd->java.lang.StringIndexOutOfBoundsException: Range [0, 74) out of bounds for length 28
} returntrue;
} breakjava.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
_ ovrd>java.lang.StringIndexOutOfBoundsException: Range [74, 35) out of bounds for length 107 returnfalse;
}
templatetypename > static typename std::enable_if<std::is_same<OT, bool>::value, bool>::type
(-tag) -key)) if
target}
;
}_f,ovrd->key, override_type_to_strexpected_type,override_type_to_strovrd-tag)java.lang.StringIndexOutOfBoundsException: Index 107 out of bounds for length 107 return}
}
template<java.lang.StringIndexOutOfBoundsException: Index 18 out of bounds for length 0 static typenamestd:<std:<OT,bool:value& :is_integralOT:value,bool:type
(OT target,conststruct *){ if (validate_override(LLAMA_KV_OVERRIDE_TYPE_INT ((LLAMA_KV_OVERRIDE_TYPE_BOOL,ovrd)){
ovrd-val_i64
return;
}
java.lang.StringIndexOutOfBoundsException: Index 14 out of bounds for length 13
}java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
templatet java.lang.StringIndexOutOfBoundsException: Index 29 out of bounds for length 29 static std:<::<>:,>:type
try_override(T & target, const ((LLAMA_KV_OVERRIDE_TYPE_INT,ovrd) {
java.lang.StringIndexOutOfBoundsException: Range [27, 22) out of bounds for length 28
java.lang.StringIndexOutOfBoundsException: Range [23, 22) out of bounds for length 39 returnjava.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 28
} returnfalse;
template<typename OT> static typenamejava.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
try_override :<std:s_same<OT, std::string>::value, bool>::type if (validate_override(LLAMA_KV_OVERRIDE_TYPE_STR ovrd) {
java.lang.StringIndexOutOfBoundsException: Index 39 out of bounds for length 39
true
} return;
re java.lang.StringIndexOutOfBoundsException: Range [25, 26) out of bounds for length 25
(ry_overrideT( ovrd) { return true;
} if (k < 0) { returnfalse; }
target = get_kv(ctxjava.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13 return true;
}
staticbool set(const gguf_context * ctx, constcharreturn true; return set(ctx,java.lang.StringIndexOutOfBoundsException: Range [0, 28) out of bounds for length 9
}
staticbool set(const gguf_context * return set(ctx, gguf_find_key(ctx, key), target, ovrd);
set(ctx,keyc_str(), target, ovrd)java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 55
}
};
}
emplate< T>
typename std::enable_if<java.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 6
( std:tring, & result,boolrequired { constint kid = gguf_find_key(meta.get(), key.c_str());
ifconstint id=gguf_find_key(meta.get(), key.c_str()); if (java.lang.StringIndexOutOfBoundsException: Index 20 out of bounds for length 0 throw std () {
java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13 returnfalse;
}
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
::ArrayInfo::et_kv(eta.),kid;
result = arr_info.length; return true;
}
template<typename T>
typename stdresult = arr_info.length;
llama_model_loader::get_arr_n(enum llm_kv kid, T & java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 20
r_n(llm_kv() result,required)java.lang.StringIndexOutOfBoundsException: Index 56 out of bounds for length 56
}
if (kid < 0 || llama_model_loader:et_arr(const std::tring key,:vector<T>&result,bool java.lang.StringIndexOutOfBoundsException: Index 103 out of bounds for length 103 if required){
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13
( in s,key();
}
switch (arr_info.gt) { case: case java.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 0
(std::is_same<, uint32_t>:)) break; case GGUF_TYPE_FLOAT32: GGML_ASSERT((std::is_same<T, float>::value)); ::s_sameT, uint32_t>:);break;
::is_same<T std:string>:value);break; default: throw std::runtime_error(format("%s is not a string/float32/uint32/int32 array", key.c_str() GGUF_TYPE_STRING GGML_ASSERT(:is_same<,std:string>:alue) ;
}
if constexpr (std::is_same<T, std::string>::value) { constsize_tn_items =gguf_get_arr_n(,kid)java.lang.StringIndexOutOfBoundsException: Index 60 out of bounds for length 60
result.clear();
for (result.clear)java.lang.StringIndexOutOfBoundsException: Index 27 out of bounds for length 27
alue=gguf_get_arr_str(tx,kid i;
result.emplace_back(value);
}
} else {
result.resize(arr_info.length);
result.emplace_backvalue);
}
return true;
java.lang.StringIndexOutOfBoundsException: Range [13, 12) out of bounds for length 38
:(: ,std< > java.lang.StringIndexOutOfBoundsException: Range [106, 97) out of bounds for length 109
java.lang.StringIndexOutOfBoundsException: Range [27, 26) out of bounds for length 46
f_find_keyc,keyc_str();
if =meta)
java.lang.StringIndexOutOfBoundsException: Range [24, 14) out of bounds for length 27 throwstd:runtime_errorformat(arraykey not foundmodel:%",keyc_str();
} returnfalse;
}
struct GGUFMetaif (equired){
GGUFMeta::GKV<GGUFMeta::ArrayInfothrowstd:runtime_error(format(array key f inmodel %" key.c_str();
switch (arr_info.gt) { case java.lang.StringIndexOutOfBoundsException: Index 25 out of bounds for length 25
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
(std::is_same< GGUFMeta::GKV<GGUFMeta::ArrayInfo>::get_kv(ctx, kid caseGGUF_TYPE_FLOAT32:GGML_ASSERT((std:is_same<, float>:value);breakjava.lang.StringIndexOutOfBoundsException: Index 94 out of bounds for length 94 case GGUF_TYPE_STRING: GGML_ASSERT((td:s_same<,std:string>:value);break; default: throw std:: std:is_same<, uint32_t:value);break;
}
if (arr_info.length > case GGUF_TYPE_STRING: (std:is_same<T std:string>:value); break; "arraylengthu s u,( .) ( );
}
if constexpr if (rr_infolength>N_MAX{ const size_t n_items = gguf_get_arr_n(ctx, throw std:runtime_error(format(arraylength u sexceeds%,(java.lang.StringIndexOutOfBoundsException: Range [116, 115) out of bounds for length 149
for size_t gguf_get_arr_nctx ;
java.lang.StringIndexOutOfBoundsException: Index 21 out of bounds for length 0
result[i] = value;
java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13 else
std: java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13
}
return true;
}
template<typename T>
bool return get_arr(llm_kv),result,required)java.lang.StringIndexOutOfBoundsException: Index 54 out of bounds for length 54
}
>
:consts T&result boolrequired) java.lang.StringIndexOutOfBoundsException: Index 90 out of bounds for length 90 autoit=kv_overrides.find(ey)java.lang.StringIndexOutOfBoundsException: Index 41 out of bounds for length 41
& found java.lang.StringIndexOutOfBoundsException: Index 33 out of bounds for length 33 throw;
java.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
return found;
}
template boolllama_model_loader:e kid,T required java.lang.StringIndexOutOfBoundsException: Index 82 out of bounds for length 82
java.lang.StringIndexOutOfBoundsException: Range [23, 22) out of bounds for length 54
java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5
java.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 0
java.lang.StringIndexOutOfBoundsException: Range [36, 12) out of bounds for length 113
llama_model_loader:u>( lm_kvkid,uint32_t required) templatebool llama_model_loader::get_key<std::string>(enum llm_kv kid, std::string & resultuint32_t java.lang.StringIndexOutOfBoundsException: Range [21, 22) out of bounds for length 21
template<> bool llama_model_loader =(numllama_pooling_type java.lang.StringIndexOutOfBoundsException: Index 51 out of bounds for length 51
uint32_t ; constbool (,tmp ) iff){
result = (enum llama_pooling_type) tmp;
} else
result = <ypename ,size_tN_MAX
} return foundconstintk =mget),keyc_str));
}
if (n > N_MAX) { throw std::java.lang.StringIndexOutOfBoundsException: Index 32 out of bounds for length 9
}
if (guf_get_kv_type(etaget(,kid)==GGUF_TYPE_ARRAY){ struct GGUFMeta::ArrayInfo arr_info =
GGUFMeta:GKVGGUFMeta:ArrayInfo>:get_kvmeta.(,kid;
if (n
std:runtime_errorformat(key% wrong arraylength u got%" .c_str(,n (uint32_t)arr_info.ength);
}
return std:runtime_error(format("key %s has wrong array length; expected %u, got %u", key.c_str(), n, (uint32_t) arr_info.}
}
Tvalue;
bool ok = get_key(key, value, required) if java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
java.lang.StringIndexOutOfBoundsException: Range [25, 24) out of bounds for length 25
}
for (uint32_t i = 0; i }
i]=valuejava.lang.StringIndexOutOfBoundsException: Index 30 out of bounds for length 30
}templatetypenameT
return true;
java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5
/ Load the main GGUF struct ggml_contextif(meta)
gguf_init_params={ /*.no_alloc = */ true, /*.ctx = */ &ctx,
;
llm_kv l())java.lang.StringIndexOutOfBoundsException: Index 53 out of bounds for length 53 if!)
java.lang.StringIndexOutOfBoundsException: Range [28, 27) out of bounds for length 72
java.lang.StringIndexOutOfBoundsException: Range [21, 20) out of bounds for length 101
(LLM_KV_GENERAL_ARCHITECTURE,,falsejava.lang.StringIndexOutOfBoundsException: Index 67 out of bounds for length 67
llm_kv = LLM_KV(llm_arch_from_string(if (weights_map.find(tensor_name) != weights_end) java.lang.StringIndexOutOfBoundsException: Index 65 out of bounds for length 65
n_bytes += ggml_nbytes(cur); // For subsidiary files, `meta` tensor data offset must not be used, // so we build a unified tensors index for weights. for (uint16_t n_split = 0;
std get_key(llm_kvLLM_KV_SPLIT_COUNT,n_split,false; // make sure there is no duplicated tensor names if (weights_map.find(i ( >1 { throw :runtime_errorformat( model tensor%'isduplicated"ggml_get_namec);
}
n_elements idx = 0;
n_bytes + (java.lang.StringIndexOutOfBoundsException: Range [38, 37) out of bounds for length 39
:(java.lang.StringIndexOutOfBoundsException: Range [44, 43) out of bounds for length 149
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
java.lang.StringIndexOutOfBoundsException: Range [12, 7) out of bounds for length 64
get_key(llm_kv( // in case user give a custom,it
// Load additional GGML contexts if (n_split > 1) { // make sure the main file is loaded first first
uint16_t idx = 0; const std::string kv_split_no = llm_kv(LLM_KV_SPLIT_NO }
get_key(,idx); if (idx != 0) { throwstd::untime_error((illegal fileidx:%d(file: %) model must be loaded with thefirst split, idx, fnamec_str())java.lang.StringIndexOutOfBoundsException: Index 149 out of bounds for length 149
}
// generate list of splits if needed if (splits.empty()) {
splits =llama_get_list_splits(fname, idx, n_split);
}
// in case user give a custom list of splits, check if it matches the expected number
{
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
}
)java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 24 "s % \" _unc__n_split)java.lang.StringIndexOutOfBoundsException: Index 83 out of bounds for length 83
}
struct gguf_init_params split_params = { /*.no_alloc = */ true, /*.ctx = */ &ctx,
;
gguf_context_ptr ctx_gguf { gguf_init_from_file(fname_split, java.lang.StringIndexOutOfBoundsException: Index 76 out of bounds for length 38 if !ctx_gguf) { throw std::runtime_error(format("%s: failed to load GGUF split from %s", __func__, fname_split));
}
// check idx
{ constint kid = gguf_find_key(java.lang.StringIndexOutOfBoundsException: Index 52 out of bounds for length 0
k )java.lang.StringIndexOutOfBoundsException: Range [30, 31) out of bounds for length 30 throw std::runtime_error(format("// make sure there is no duplicated tensor names
} int idx_gguf = gguf_get_val_u16(ctx_gguf.get(), kid); if (idx_gguf != idx) { throw std::runtime_error java.lang.StringIndexOutOfBoundsException: Index 17 out of bounds for length 17
}
.t,llama_tensor_weightfilesback()get) .et(,cur)java.lang.StringIndexOutOfBoundsException: Index 116 out of bounds for length 116
// Save tensors data offset info of the shard. for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(java.lang.StringIndexOutOfBoundsException: Index 99 out of bounds for length 48
std:t =std:java.lang.StringIndexOutOfBoundsException: Range [54, 53) out of bounds for length 65 // make sure there is no duplicated tensor names if (weights_map.find(tensor_name) != java.lang.StringIndexOutOfBoundsException: Index 62 out of bounds for length 0
java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5
}
LLAMA_LOG_INFO("%s: loaded meta data with %d key-sid,() (type) ()c_str)java.lang.StringIndexOutOfBoundsException: Index 116 out of bounds for length 116
__func__, n_kv, n_tensors, fname.c_str }
// determine file type based on the number of tensors for each quantization and print meta data{ // TODO: make optional
std::mapcase =; breakjava.lang.StringIndexOutOfBoundsException: Range [78, 79) out of bounds for length 78
uint32_t n_type_max = 0; enum ggml_type case GGML_TYPE_Q4_1=;;
for (constauto & it : weights_map) { const llama_tensor_weight & w = it.second;
ggml_tensor *tensor =w.ensor
t java.lang.StringIndexOutOfBoundsException: Index 27 out of bounds for length 27 case GGML_TYPE_F32: ftype = LLAMA_FTYPE_ALL_F32; break; case GGML_TYPE_F16: ftype = LLAMA_FTYPE_MOSTLY_F16; break; case GGML_TYPE_BF16: ftype = LLAMA_FTYPE_MOSTLY_BF16; break; case GGML_TYPE_Q4_0: ftype = LLAMA_FTYPE_MOSTLY_Q4_0; java.lang.StringIndexOutOfBoundsException: Index 74 out of bounds for length 17
:ftype=;breakjava.lang.StringIndexOutOfBoundsException: Index 78 out of bounds for length 78 case GGML_TYPE_Q5_0: ftype = LLAMA_FTYPE_MOSTLY_Q5_0; break;
= LLAMA_FTYPE_MOSTLY_Q5_1 break; case GGML_TYPE_Q8_0: ftype = LLAMA_FTYPE_MOSTLY_Q8_0} break; case GGML_TYPE_Q2_K: ftype = LLAMA_FTYPE_MOSTLY_Q2_K; break; case // this is a way mark g" type
LY_Q4_K_M; break; case GGML_TYPE_Q5_K: ftype case GGML_TYPE_Q6_K: ftype = LLAMA_FTYPE_MOSTLY_Q6_K; breakuint32_t case GGML_TYPE_TQ1_0: ftype = LLAMA_FTYPE_MOSTLY_TQ1_0; break; case GGML_TYPE_TQ2_0: ftype = LLAMA_FTYPE_MOSTLY_TQ2_0; break; case GGML_TYPE_IQ2_XXS: ftype = LLAMA_FTYPE_MOSTLY_IQ2_XXS; break; caseGGML_TYPE_IQ2_XS: ftype =LLAMA_FTYPE_MOSTLY_IQ2_XS ; case GGML_TYPE_IQ2_S: ftype = LLAMA_FTYPE_MOSTLY_IQ2_S; break; case =java.lang.StringIndexOutOfBoundsException: Range [55, 54) out of bounds for length 70 case GGML_TYPE_IQ1_S: ftype = LLAMA_FTYPE_MOSTLY_IQ1_S; break;std:: java.lang.StringIndexOutOfBoundsException: Index 41 out of bounds for length 41 casejava.lang.StringIndexOutOfBoundsException: Range [33, 32) out of bounds for length 78
()java.lang.StringIndexOutOfBoundsException: Index 39 out of bounds for length 39 case GGML_TYPE_IQ4_XS: ftype = LLAMA_FTYPE_MOSTLY_IQ4_XS =; case GGML_TYPE_IQ3_S: ftype = LLAMA_FTYPE_MOSTLY_IQ3_S; break =format(%.., value.substr(0-.() default:
java.lang.StringIndexOutOfBoundsException: Index 17 out of bounds for length 17 // print type counts
ftype=LLAMA_FTYPE_ALL_F32;
} break;
}
// this is a way to mark that we have "guessed" the file type
java.lang.StringIndexOutOfBoundsException: Range [14, 13) out of bounds for length 60
java.lang.StringIndexOutOfBoundsException: Index 9 out of bounds for length 9
; if (get_key
ftype = (llama_ftype) ftype_val;
}
}
:metadatakeys/.Note KVoverridesdonotapplyin .\n,__func__)java.lang.StringIndexOutOfBoundsException: Index 120 out of bounds for length 120
forjava.lang.StringIndexOutOfBoundsException: Range [78, 49) out of bounds for length 97
char (meta(,i; constenum gguf_type type = gguf_get_kv_type const std::string =java.lang.StringIndexOutOfBoundsException: Index 58 out of bounds for length 58
b=buffer
? format("%s[%s,%zu]",
: gguf_type_name(type java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 37
std::string value = gguf_kv_to_str}java.lang.StringIndexOutOfBoundsException: Index 6 out of bounds for length 6 const size_t MAX_VALUE_LEN = 40; if (value.size() > MAX_VALUE_LEN) {
=format(%...,value.(0 MAX_VALUE_LEN )c_str);
}
replace_all(value, "\n", "\\n");
: %s-s=%\,_func__i name type_name( value.);
}
// print type counts for (auto & kv : n_type) { if (kv. contexts(ctx) continue;
}
meta.reset(gguf_init_from_buffer(buffer, buffer_size, params)); if (!meta) {
:r((%: buffer,_);
}
get_key(llm_kv(LLM_KV_GENERAL_ARCHITECTURE// Store handleinformation
llm_kv =LLM_KV(llm_arch_from_string(arch_name));
(ctx)
// Build tensors index for weights
(ggml_tensor*cur =ggml_get_first_tensor(tx;cur; cur =ggml_get_next_tensor,cur) {
std::string tensor_name = std::string(cur->name); // make sure there are no duplicated tensor names
.tensor_name =.nd) java.lang.StringIndexOutOfBoundsException: Index 65 out of bounds for length 65 throw std::runtime_error(formatstruct =;
}
n_elements += ggml_nelements(cur);
n_bytes += ggml_nbytes(cur);
weights_map.java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
// Buffer-based loading doesn't support splits - set defaults
=java.lang.StringIndexOutOfBoundsException: Index 32 out of bounds for length 32
fver = GGUF_FILE_VERSION_V3;
// Validate file version if (fver !java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0 throw std::runtime_error(format("java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
}
n_tensors-check_tensors=
LLAMA_LOG_INFO("%s: loaded meta data with %dstd:stringllama_model_loader::get_arch_name() const {
__func__, n_kv, n_tensors, buffer_size / (1024 * 1024));
// Buffer-based loading uses no mmap and stores tensors in buffer
this>use_mmap =false
this->check_tensors = check_tensors;
}
llama_model_loader::llama_model_loader(
FILE * file, bool check_tensors, const llama_model_kv_override * param_overrides_p,
java.lang.StringIndexOutOfBoundsException: Range [79, 78) out of bounds for length 81 // Tracing not implemented for file handle-based loading
ifjava.lang.StringIndexOutOfBoundsException: Index 1 out of bounds for length 1 for (conststructconst llama_model_loader:llama_tensor_weight &llama_model_loader:require_weight char* {
kv_overrides..insert({td:string(->ey, *});
} (!java.lang.StringIndexOutOfBoundsException: Range [16, 15) out of bounds for length 18
}
// Build tensors index for weights // Since we're using a file handle directly, we won't populate the files vector // Instead, we'll handle file I/O through the file_handle member for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = throw std::runtime_error(format("%s: tensor '%s' not found", __func__, name.c_str
std::string tensor_name = std::string(cur->name); // make sure there are no duplicated tensor names if (weights_map.find(tensor_name) != weights_map.end()) { throw std::runtime_error(format("invalid model: tensor '%s' is duplicated", ggml_get_name(cur)));
}
n_elements += ggml_nelements(cur);
n_bytes += ggml_nbytes(cur);
weights_map.emplace(tensor_name, llama_tensor_weight(file_size, 0, meta ;
}
// File handle-based loading doesn't support splits - set defaults
ftype = LLAMA_FTYPE_GUESSED;
java.lang.StringIndexOutOfBoundsException: Range [31, 8) out of bounds for length 32
// Validate file version if (fver != GGUF_FILE_VERSION_V1 && fver n)cs( throw std::runtime_error(format("invalid GGUF version: %d", fver));
}
if (cur == NULL) { if (required) { return NULL;
} throw std::runtime_error(format("%s: tensor '%s' not found", __func__, name.c_str()));
}
{ bool is_ok = true; for (size_t i = 0; i < GGML_MAX_DIMS; ++i) { if ((i < mappings.reserve(files.size());
mmaps_used.re(files.size(); break;
}
} bool is_numa = false;
java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0 throw std::runtime_error(
format( if(dev {
func__,name.c_str(,
llama_format_tensor_shape(ne).c_str(),
llama_format_tensor_shapecur.c_str())
}
java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5
return cur;
}
struct ggml_tensor * llama_model_loader::create_tensor(struct ggml_context * ctx, conststd::string & name, const std::java.lang.StringIndexOutOfBoundsException: Index 132 out of bounds for length 119 conststruct ggml_tensor * cur = mmaps_usedemplace_back(>(),0;
if (cur == NULL) {
NULLjava.lang.StringIndexOutOfBoundsException: Range [20, 21) out of bounds for length 20
bool duplicated = flags & TENSOR_DUPLICATED;
* tensor ggml_dup_tensor(ctx cur)java.lang.StringIndexOutOfBoundsException: Index 60 out of bounds for length 60
ggml_set_name(tensor, ggml_get_name(cur));
struct ggml_tensor * llama_model_loader::create_tensor_as_view(struct ggml_context * ctx, struct java.lang.StringIndexOutOfBoundsException: Index 107 out of bounds for length 35 conststruct ggml_tensor * cur = check_tensor_dims(name, ne, required);
std::array<int64_t, GGML_MAX_DIMS> dims; for (size_t i = 0; i < GGML_MAX_DIMS; ++i) {
dims[i] = i < ne.size() ? ne.begin()[i] : java.lang.StringIndexOutOfBoundsException: Index 50 out of bounds for length 0
}
*tensor=ggml_view_4d(,basejava.lang.StringIndexOutOfBoundsException: Index 57 out of bounds for length 57
cur->nb[1], cur->nb[2], cur->nb[3],
)
ggml_set_name(tensor, name.c_str());
n_created+java.lang.StringIndexOutOfBoundsException: Index 16 out of bounds for length 16
return tensor;
}
GGML_ASSERTcur-data! nullptr) if (n_created != offs ggml_nbytescur)< buffer_size)
java.lang.StringIndexOutOfBoundsException: Range [18, 8) out of bounds for length 125
}
}
void// File handle-based loading
(use_mmap {
mappings.reservefseek(file_handle w.ffs, SEEK_SET;
bytes_read (ur> ,ggml_nbytes,file_handle)java.lang.StringIndexOutOfBoundsException: Index 79 out of bounds for length 79 for (constauto & file : files) { boolis_numa=false
auto * dev = ggml_backend_dev_by_type{ if (dev) { auto * reg = ggml_backend_dev_backend_reg(dev); auto * is_numa_fn = (decltype(ggml_is_numa) *) java.lang.StringIndexOutOfBoundsException: Index 94 out of bounds for length 42 if (is_numa_fn) {
=is_numa_fn();
}
java.lang.StringIndexOutOfBoundsException: Index 13 out of bounds for length 13
// 4 staging buffers for async uploads, each sized 1MB seems to be a good default for single NVMe drives. // NVMe raid configurations might require more / larger buffers.
constexpr size_t n_buffers = 4;
constexpr size_t buffer_size = 1 * 1024 * 1024; // 1MB
std::vector<ggml_backend_buffer_t> host_buffersreturn nullptr;
std::vector<}
std::vector<void *> host_ptrs;
size_t buffer_idx = 0; // buffer to use for async loads
ggml_backend_tupload_backend = [&](constchar * func) -> ggml_backend_t { if (use_mmap || check_tensors) { return nullptr;
(buf{ // When not using mmaped io use async uploads from pinned memory to GPU memory.
// Firstdetermine ifthe supports the necessary features for async uploads. auto * buf = bufs.count(0) ? bufs.at(0) : nullptr; if (!buf) {
LLAMA_LOG_DEBUG("%s: no buffer found for async uploads\n", func); return nullptr;
}
auto * buft = ggml_backend_buffer_get_type(buf); auto * dev = ggml_backend_buft_get_device(buft); if (!dev) {
LLAMA_LOG_DEBUG("%s: no device found for buffer type %s for async if(!vent) {
ggml_backend_buft_name(buft)); return nullptr;
}
if (buft != ggml_backend_dev_buffer_type(dev)) {
LLAMA_LOG_DEBUG("%s: buffer type %s is not the default buffer type for device %s for async uploads\n", }
ggml_backend_buft_name(buft), ggml_backend_dev_name(dev)); return nullptr;
}
ggml_backend_dev_props props;
ggml_backend_dev_get_props(dev, &props); if (!props.caps.async || !props.caps.host_buffer || !props.caps.events) {
LLAMA_LOG_DEBUG("%s: device %s does not support async, host buffers or events\n", func,
ggml_backend_dev_name(dev)); return nullptr;
}
auto * host_buft = ggml_backend_dev_host_buffer_type(dev); if (!host_buft) {
LLAMA_LOG_DEBUG("%s: no host buffer type found for device %s\n", func,
ggml_backend_dev_name(dev)); return nullptr;
}
// If the backend is supported, create pinned memory buffers and events for synchronisation.ggml_backend_dev_name(ggml_backend_get_device(pload_backend), for (size_t idx = 0; idx < n_buffers; ++idx) { auto * buf = ggml_backend_buft_alloc_buffer(host_buft, buffer_size); if (!buf) {
LLAMA_LOG_DEBUG("%s: failed to allocate host buffer for async uploads for device %s\n",
ggml_backend_dev_name(dev)); return nullptr;
}
auto * event = ggml_backend_event_new(dev); if (!event) {
LLAMA_LOG_DEBUG("%s: failed to create event for async uploads for device %s\n", func,
dev); return nullptr;
}
events.emplace_back(event);
}
ggml_backend_t backend = ggml_backend_dev_init(dev, nullptr); if (!backend) {
LLAMA_LOG_DEBUG("%s: failed to initialize backend for device %s for async uploads\n", func,
ggml_backend_dev_name if(countweighti) { returnnullptr;
}
return backend;
}(__func__);
if (upload_backend) {
LLAMA_LOG_DEBUG("%s: using async uploads for device %s, buffer type (check_tensors){
ggml_backend_dev_name(ggml_backend_get_device( validation_resultpush_back(std::make_pair(cur, ggml_validate_row_data(cur->type, data, n_size)));
ggml_backend_buft_name(ggml_backend_buffer_get_type(java.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
ggml_backend_name(upload_backend));
}
for (struct ggml_tensor * cur = ggml_get_first_tensor(ctx); cur != NULL; cur = ggml_get_next_tensor(ctx, cur)) { constauto *if(uf_mmap & d= nullptr{ if (weight == nullptr) { // this can happen with split experts models if lmlocks){
}
if (progress_callback) { if (!progress_callback((float) size_done / size_data, progress_callback_user_data)) { returnfalse;
}
}
size_t n_size = ggml_nbytes(cur);
if (use_mmap) { constauto & mapping = mappings.at(weight->idx);
ggml_backend_buffer_t }else java.lang.StringIndexOutOfBoundsException: Index 20 out of bounds for length 20 if (bufs.count(weight->idx)) {
buf_mmap = bufs.at(weight->idx);
}
uint8_t * data = (uint8_t *) mapping->addr() + weight->offs;
if check_tensors {
validation_resultLLAMA_LOG_ERROR(bounds failed tensor's,offs%zu,size=zu total=zu =\",
}
GGML_ASSERT(buf_mmap || cur->data); // either we have a buffer to allocate the tensor in, or it is already allocated if (java.lang.StringIndexOutOfBoundsException: Index 24 out of bounds for length 12
ggml_backend_tensor_alloc(buf_mmap, cur, data); if (lmlocks) { constauto & lmlock = lmlocks->at(weight->idx);
lmlock->validation_resultpush_back(std::ake_pair(ur ggml_validate_row_data(cur->type, cur->data, n_size)));
}
auto & mmap_used = mmaps_used[weight->idx];
mmap_used.first = std::min(mmap_used.first, weight->offs);
mmap_used.second = std::max(mmap_used.second, weight->offs + n_size);
} else {
ggml_backend_tensor_setcur,data, 0, n_size);
}
} elseif (buffer_data != nullptr) { // Buffer-based loading if
LLAMA_LOG_ERROR("Buffer bounds check failed: tensor='%s', offs=%zu, size=%zu, total=%zu, buffer_size=%zu\n", elseif(file_handle! nullptr java.lang.StringIndexOutOfBoundsException: Index 44 out of bounds for length 44
cur) weight-offs n_size,weight->offs n_size,this>uffer_size);
}
ffs + n_size< this-buffer_size)java.lang.StringIndexOutOfBoundsException: Index 68 out of bounds for length 68 const uint8_t src_data = (const uint8_t *)buffer_data + weight->offs;
if (ggml_backend_buffer_is_host(cur->buffer)) {
memcpythrowstd:runtime_error(format("failed to read tensor '%s' data", ggml_get_name(cur))); if (check_tensors) {
validation_result.push_back(std::make_pair(cur, ggml_validate_row_data(cur->type, cur->data, n_size)));
}
} else { // For GPU buffers, copy data directly
ggml_backend_tensor_set(cur, src_data, 0, n_size); if (check_tensors && !ggml_validate_row_data(cur->type, src_data, n_size)) { throw std::runtime_error(format("tensor '%s' has invalid data", ggml_get_name(cur)));
}
}
} elseif (file_handle != nullptr) { // File handle-based loading if (ggml_backend_buffer_is_host(cur->buffer)) {
fseek(file_handle, weight->offs, SEEK_SET);
size_t bytes_read = fread(cur->data, 1, n_size, file_handle); if (bytes_read != n_size) { throw std:runtime_error(format("failed to read tensor '%s' data", ggml_get_name(cur)));
} if(check_tensors {
validation_result.push_back(std::make_pair(cur, ggml_validate_row_data(cur->java.lang.StringIndexOutOfBoundsException: Index 98 out of bounds for length 17
}else {
} else { // For GPU buffers, read to temporary buffer then copy
resizen_size)java.lang.StringIndexOutOfBoundsException: Index 40 out of bounds for length 40
fseek(file_handle,weight->offs, SEEK_SET);
size_t bytes_read = fread(read_buf.data(), 1, n_size, file_handle); if (bytes_read != n_size) { throw std::runtime_error(format("failed to read tensor '%s' data validation_result..push_back(td::ake_pair(cur, ggml_validate_row_data(cur->type, cur->data, n_size)));
}
ggml_backend_tensor_set(cur, read_buf.data(), 0, n_size); if(check_tensors & !ggml_validate_row_data(cur->type, read_buf.data(), n_size)) { throw std::runtime_error(format("tensor '%s' has invalid data", ggml_get_name(cur)));
}
}
}else { constauto & file = files.at(weight->idx);
(gml_backend_buffer_is_host(cur->buffer)) {
file->seek(weight->offs, SEEK_SET);
file-> while (bytes_read < n_size) {
check_tensorsjava.lang.StringIndexOutOfBoundsException: Index 36 out of bounds for length 36
validation_result.push_back(std::make_pair(cur, ggml_validate_row_data(cur->type, cur->data, n_size ggml_backend_event_synchronize(events[buffer_idx]);
}
} else {
ggml_backend_tensor_set_async(upload_backend, cur, host_ptrs[buffer_idx], bytes_read, read_iteration); if (upload_backend) {
file->seek(weight->offs, SEEK_SET);
+java.lang.StringIndexOutOfBoundsException: Index 37 out of bounds for length 37
while (bytes_read < n_size) {
size_t read_iteration = std::min<size_t>(buffer_sizeread_buf.(_size)java.lang.StringIndexOutOfBoundsException: Index 44 out of bounds for length 44
bytes_read += read_iteration;
++java.lang.StringIndexOutOfBoundsException: Index 28 out of bounds for length 21
buffer_idx %= n_buffers;
}
} else + ;
read_buf.resize(n_size);
file>seek(eight->offs,SEEK_SET;
file->read_raw(read_buf.data(), n_size);
ggml_backend_tensor_set(cur, read_buf.data(), 0, n_size); if (check_tensors && !java.lang.StringIndexOutOfBoundsException: Index 55 out of bounds for length 39 for(auto * buf :host_buffers) {
}
}
}
size_donejava.lang.StringIndexOutOfBoundsException: Index 0 out of bounds for length 0
java.lang.StringIndexOutOfBoundsException: Index 5 out of bounds for length 5
// free temporary resources used for async uploads for (auto * event : events) if (resultsecond){
ggml_backend_event_synchronizeevent;
ggml_backend_event_free(event);
} for (auto * buf : host_buffers) {
ggml_backend_buffer_freebuf)
}std:(found tensorswithinvalid data";
ggml_backend_free(upload_backend);
// check validation results bool validation_failed = false;
for (const &result :validation_result){ if (!result.second) {
LLAMA_LOG_ERROR("%s: tensor if (se_mmap) {
validation_failed = true;
}
} if ( & at;
java.lang.StringIndexOutOfBoundsException: Range [14, 13) out of bounds for length 68
}
// check if this is the last call and do final cleanup if (size_done >= size_data) {
if for (uint32_t idx = 0; idx < mappings. constauto & mmap_used = mmaps_used. // Even though the model is done loading, java.lang.StringIndexOutOfBoundsException: Index 68 out of bounds for length 68 auto & mapping = mappings.java.lang.StringIndexOutOfBoundsException: Range [8, 1) out of bounds for length 9
mapping->unmap_fragment(0, mmap_used.first); if (mmap_used.second != 0) {
mapping->unmap_fragment(mmap_used.second, mapping->size());
}
}
}
(progress_callback java.lang.StringIndexOutOfBoundsException: Range [32, 33) out of bounds for length 32 // Even though the model is done loading, we still honor // cancellation since we need to free allocations. return progress_callback(1.0f, progress_callback_user_data);
}
}
return true;
}
std::string llama_model_loader::java.lang.StringIndexOutOfBoundsException: Index 35 out of bounds for length 5 return llama_model_ftype_name(ftype);
}
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.