@@ -863,9 +863,6 @@ async def _fetch_and_merge(repo_id: Optional[str]):
863863 metadata ["tokenizer" ] = tokenizer_json
864864
865865 await _fetch_and_merge (huggingface_id )
866- if huggingface_id and huggingface_id .lower ().endswith ("-gguf" ):
867- base_repo = huggingface_id [:- 5 ]
868- await _fetch_and_merge (base_repo )
869866
870867 try :
871868 layer_info = get_model_layer_info (file_path ) or {}
@@ -1210,6 +1207,20 @@ async def process_model(model):
12101207 if result is not None :
12111208 valid_results .append (result )
12121209
1210+ if model_format == "gguf" :
1211+ def _gguf_sort_key (item : Dict [str , Any ]):
1212+ quantizations = item .get ("quantizations" ) or {}
1213+ size_candidates = [
1214+ q .get ("total_size" ) or 0
1215+ for q in quantizations .values ()
1216+ if isinstance (q , dict )
1217+ ]
1218+ positive_sizes = [size for size in size_candidates if size > 0 ]
1219+ min_size = min (positive_sizes ) if positive_sizes else float ("inf" )
1220+ return (min_size , - (item .get ("downloads" ) or 0 ), item .get ("id" ) or "" )
1221+
1222+ valid_results .sort (key = _gguf_sort_key )
1223+
12131224 return valid_results [:limit ]
12141225
12151226
@@ -1219,27 +1230,33 @@ async def _process_single_model(model, model_format: str) -> Optional[Dict]:
12191230 logger .info (f"Processing model: { model .id } " )
12201231
12211232 quantizations : Dict [str , Dict ] = {}
1233+ mmproj_files : List [Dict [str , Any ]] = []
12221234 safetensors_files : List [Dict ] = []
12231235 repo_files : List [Dict [str , Any ]] = []
12241236
12251237 if hasattr (model , "siblings" ) and model .siblings :
12261238 if model_format == "gguf" :
1227- # Group GGUF files by logical quantization, handling multi-part shards
1228- # Accept both plain `.gguf` and multi-part patterns like `.gguf.part1of2`
1229- # Exclude mmproj (vision/multimodal projection) files – they are extensions, not standalone quants
1239+ # Group GGUF files by logical quantization, handling multi-part shards.
12301240 gguf_siblings = [
12311241 s
12321242 for s in model .siblings
12331243 if isinstance (getattr (s , "rfilename" , None ), str )
12341244 and re .search (r"\.gguf(\.|$)" , s .rfilename )
1235- and "mmproj" not in s .rfilename .lower ()
12361245 ]
12371246 logger .info (f"Model { model .id } : { len (gguf_siblings )} GGUF files found" )
12381247 if not gguf_siblings :
12391248 return None
12401249
12411250 for sibling in gguf_siblings :
12421251 filename = sibling .rfilename
1252+ if "mmproj" in filename .lower ():
1253+ mmproj_files .append (
1254+ {
1255+ "filename" : filename ,
1256+ "size" : getattr (sibling , "size" , 0 ) or 0 ,
1257+ }
1258+ )
1259+ continue
12431260 # Normalize filename by stripping shard suffix patterns like:
12441261 # -00001-of-00002.gguf (TheBloke-style)
12451262 # .gguf.part1of2 (Hugging Face-style multi-part)
@@ -1298,25 +1315,9 @@ async def _process_single_model(model, model_format: str) -> Optional[Dict]:
12981315 else 0.0
12991316 )
13001317
1301- # Siblings from list_models often have size=None; fetch accurate sizes from Hub
1302- try :
1303- all_filenames = [s .rfilename for s in gguf_siblings ]
1304- accurate_sizes = get_accurate_file_sizes (model .id , all_filenames )
1305- if accurate_sizes :
1306- for entry in quantizations .values ():
1307- for f in entry ["files" ]:
1308- f ["size" ] = accurate_sizes .get (f ["filename" ]) or f ["size" ] or 0
1309- entry ["total_size" ] = sum (f ["size" ] for f in entry ["files" ])
1310- entry ["size_mb" ] = (
1311- round (entry ["total_size" ] / (1024 * 1024 ), 2 )
1312- if entry ["total_size" ]
1313- else 0.0
1314- )
1315- except Exception as size_err :
1316- logger .debug (f"Could not fetch accurate sizes for { model .id } : { size_err } " )
1317-
1318- # If no quantizations were detected after grouping, skip this model
1319- if not quantizations :
1318+ # Search should stay to a single HF API call. Accurate file sizes are lazy-loaded on expand.
1319+ # If no downloadable GGUF entries were detected after grouping, skip this model.
1320+ if not quantizations and not mmproj_files :
13201321 return None
13211322 else :
13221323 safetensors_files = []
@@ -1338,15 +1339,6 @@ async def _process_single_model(model, model_format: str) -> Optional[Dict]:
13381339 )
13391340 if not safetensors_files :
13401341 return None
1341- # Fetch accurate sizes; list_models siblings often have size=None
1342- try :
1343- st_filenames = [f ["filename" ] for f in safetensors_files ]
1344- accurate_sizes = get_accurate_file_sizes (model .id , st_filenames )
1345- if accurate_sizes :
1346- for f in safetensors_files :
1347- f ["size" ] = accurate_sizes .get (f ["filename" ]) or 0
1348- except Exception as size_err :
1349- logger .debug (f"Could not fetch accurate sizes for { model .id } : { size_err } " )
13501342 else :
13511343 return None
13521344
@@ -1364,6 +1356,7 @@ async def _process_single_model(model, model_format: str) -> Optional[Dict]:
13641356 "tags" : model .tags or [],
13651357 "model_format" : model_format ,
13661358 "quantizations" : quantizations if model_format == "gguf" else {},
1359+ "mmproj_files" : mmproj_files if model_format == "gguf" else [],
13671360 "safetensors_files" : (
13681361 safetensors_files if model_format == "safetensors" else []
13691362 ),
@@ -1668,7 +1661,7 @@ async def get_model_details(model_id: str) -> Dict:
16681661 config_path = hf_hub_download (
16691662 repo_id = model_id ,
16701663 filename = "config.json" ,
1671- local_dir = "data/temp " ,
1664+ local_dir = "data/hf-cache " ,
16721665 local_dir_use_symlinks = False ,
16731666 )
16741667
0 commit comments