Compare commits

...
Author SHA1 Message Date
Xuan Son Nguyen b4e5d8b616 rm pocket-tts 2026-08-17 10:14:14 +02:00
Xuan Son Nguyen 9edb2d2dba BailingMoeV3ForCausalLM 2026-08-17 10:11:26 +02:00
Xuan Son Nguyen 507ace2827 Merge branch 'master' into xsn/convert_add_example 2026-08-17 10:09:07 +02:00
Xuan Son Nguyen 6e97dc0518 add more variants 2026-08-17 09:12:19 +02:00
Xuan Son Nguyen 584b4f93c6 add docs 2026-08-17 02:36:33 +02:00
Xuan Son Nguyen 71c7866a16 convert: add @ModelBase.example 2026-08-17 02:31:10 +02:00
86 changed files with 236 additions and 0 deletions
+1
View File
@@ -13,6 +13,7 @@ from .llama import LlamaModel
@ModelBase.register("AfmoeForCausalLM")
@ModelBase.example("arcee-ai/Trinity-Large-Thinking")
class AfmoeModel(LlamaModel):
model_arch = gguf.MODEL_ARCH.AFMOE
+1
View File
@@ -16,6 +16,7 @@ from .llama import LlamaModel
@ModelBase.register("ArcticForCausalLM")
@ModelBase.example("Snowflake/snowflake-arctic-instruct")
class ArcticModel(TextModel):
model_arch = gguf.MODEL_ARCH.ARCTIC
+1
View File
@@ -9,6 +9,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("BaichuanForCausalLM", "BaiChuanForCausalLM")
@ModelBase.example("baichuan-inc/Baichuan2-7B-Chat", "baichuan-inc/Baichuan-7B")
class BaichuanModel(TextModel):
model_arch = gguf.MODEL_ARCH.BAICHUAN
+3
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("BailingMoeForCausalLM")
@ModelBase.example("inclusionAI/Ling-lite")
class BailingMoeModel(TextModel):
model_arch = gguf.MODEL_ARCH.BAILINGMOE
@@ -108,6 +109,7 @@ class BailingMoeModel(TextModel):
@ModelBase.register("BailingMoeV2ForCausalLM")
@ModelBase.example("inclusionAI/Ling-mini-2.0")
class BailingMoeV2Model(TextModel):
model_arch = gguf.MODEL_ARCH.BAILINGMOE2
@@ -189,6 +191,7 @@ class BailingMoeV2Model(TextModel):
@ModelBase.register("SarvamMoEForCausalLM", "modeling_sarvam_moe.SarvamMoEForCausalLM")
@ModelBase.example("sarvamai/sarvam-30b")
class SarvamMoEModel(BailingMoeV2Model):
model_arch = gguf.MODEL_ARCH.BAILINGMOE2
# Sarvam-MoE shares the BailingMoeV2 architecture; only differences:
+1
View File
@@ -13,6 +13,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("BailingMoeV3ForCausalLM")
@ModelBase.example("inclusionAI/Ling-3.0-tiny", "inclusionAI/Ling-3.0-flash")
class BailingMoeV3Model(TextModel):
model_arch = gguf.MODEL_ARCH.BAILINGMOE3
supports_mtp_export = True
+8
View File
@@ -1149,6 +1149,14 @@ class ModelBase:
return modelcls
return func
@classmethod
def example(cls, *hf_repos: str) -> Callable[[AnyModel], AnyModel]:
del hf_repos # unused
def func(modelcls: AnyModel) -> AnyModel:
return modelcls
return func
@classmethod
def print_registered_models(cls):
for model_type, model_classes in cls._model_classes.items():
+9
View File
@@ -15,6 +15,7 @@ from .base import ModelBase, SentencePieceTokenTypes, TextModel, gguf, logger
@ModelBase.register("BertModel", "BertForMaskedLM", "CamembertModel", "BertForSequenceClassification")
@ModelBase.example("BAAI/bge-small-en-v1.5", "dangvantuan/sentence-camembert-base")
class BertModel(TextModel):
model_arch = gguf.MODEL_ARCH.BERT
@@ -240,6 +241,7 @@ class BertModel(TextModel):
@ModelBase.register("DistilBertModel", "DistilBertForMaskedLM", "DistilBertForSequenceClassification")
@ModelBase.example("distilbert/distilbert-base-uncased")
class DistilBertModel(BertModel):
model_arch = gguf.MODEL_ARCH.BERT
@@ -263,6 +265,7 @@ class DistilBertModel(BertModel):
@ModelBase.register("RobertaModel", "RobertaForSequenceClassification")
@ModelBase.example("sentence-transformers/stsb-roberta-base")
class RobertaModel(BertModel):
model_arch = gguf.MODEL_ARCH.BERT
@@ -312,6 +315,7 @@ class RobertaModel(BertModel):
@ModelBase.register("NomicBertModel")
@ModelBase.example("nomic-ai/nomic-embed-text-v1.5")
class NomicBertModel(BertModel):
model_arch = gguf.MODEL_ARCH.BERT
@@ -400,6 +404,7 @@ class NomicBertModel(BertModel):
@ModelBase.register("NeoBERT", "NeoBERTLMHead", "NeoBERTForSequenceClassification")
@ModelBase.example("chandar-lab/NeoBERT")
class NeoBert(BertModel):
model_arch = gguf.MODEL_ARCH.NEO_BERT
@@ -431,6 +436,7 @@ class NeoBert(BertModel):
@ModelBase.register("EuroBertModel", "JinaEmbeddingsV5Model")
@ModelBase.example("hf-tiny-v2/tiny-random-EuroBertModel", "jinaai/jina-embeddings-v5-text-nano")
class EuroBertModel(TextModel):
model_arch = gguf.MODEL_ARCH.EUROBERT
@@ -459,6 +465,7 @@ class EuroBertModel(TextModel):
@ModelBase.register("XLMRobertaModel", "XLMRobertaForSequenceClassification")
@ModelBase.example("BAAI/bge-m3")
class XLMRobertaModel(BertModel):
model_arch = gguf.MODEL_ARCH.BERT
_lora_files = {}
@@ -561,6 +568,7 @@ class XLMRobertaModel(BertModel):
@ModelBase.register("JinaBertModel", "JinaBertForMaskedLM")
@ModelBase.example("jinaai/jina-embeddings-v2-base-en")
class JinaBertV2Model(BertModel):
model_arch = gguf.MODEL_ARCH.JINA_BERT_V2
@@ -588,6 +596,7 @@ class JinaBertV2Model(BertModel):
@ModelBase.register("ModernBertModel", "ModernBertForMaskedLM", "ModernBertForSequenceClassification")
@ModelBase.example("answerdotai/ModernBERT-base")
class ModernBertModel(BertModel):
model_arch = gguf.MODEL_ARCH.MODERN_BERT
+1
View File
@@ -9,6 +9,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("BitnetForCausalLM", "BitNetForCausalLM")
@ModelBase.example("microsoft/bitnet-b1.58-2B-4T")
class BitnetModel(TextModel):
model_arch = gguf.MODEL_ARCH.BITNET
+1
View File
@@ -13,6 +13,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("BloomForCausalLM", "BloomModel")
@ModelBase.example("bigscience/bloom-560m")
class BloomModel(TextModel):
model_arch = gguf.MODEL_ARCH.BLOOM
+2
View File
@@ -12,6 +12,8 @@ from .llama import LlamaModel
@ModelBase.register("ChameleonForConditionalGeneration")
@ModelBase.register("ChameleonForCausalLM") # obsolete
# [TAG_HF_EXAMPLE_GATED] facebook/chameleon-7b is gated
# [TAG_HF_EXAMPLE_MISSING]
class ChameleonModel(TextModel):
model_arch = gguf.MODEL_ARCH.CHAMELEON
+1
View File
@@ -9,6 +9,7 @@ from .base import ModelBase, SentencePieceTokenTypes, TextModel, gguf
@ModelBase.register("GlmForCausalLM", "ChatGLMModel", "ChatGLMForConditionalGeneration")
@ModelBase.example("THUDM/chatglm3-6b", "zai-org/glm-4-9b-chat-hf")
class ChatGLMModel(TextModel):
model_arch = gguf.MODEL_ARCH.CHATGLM
+1
View File
@@ -4,6 +4,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("CodeShellForCausalLM")
@ModelBase.example("WisdomShell/CodeShell-7B")
class CodeShellModel(TextModel):
model_arch = gguf.MODEL_ARCH.CODESHELL
+2
View File
@@ -11,6 +11,7 @@ from .llama import LlamaModel
@ModelBase.register("CogVLMForCausalLM")
@ModelBase.example("THUDM/cogvlm2-llama3-chat-19B", "THUDM/cogvlm-chat-hf")
class CogVLMVisionModel(MmprojModel):
def set_gguf_parameters(self):
@@ -29,5 +30,6 @@ class CogVLMVisionModel(MmprojModel):
@ModelBase.register("CogVLMForCausalLM")
@ModelBase.example("THUDM/cogvlm2-llama3-chat-19B", "THUDM/cogvlm-chat-hf")
class CogVLMModel(LlamaModel):
model_arch = gguf.MODEL_ARCH.COGVLM
+5
View File
@@ -12,6 +12,8 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("CohereForCausalLM")
# [TAG_HF_EXAMPLE_GATED] CohereLabs/c4ai-command-r-v01 is gated
# [TAG_HF_EXAMPLE_MISSING]
class CommandR2Model(TextModel):
model_arch = gguf.MODEL_ARCH.COMMAND_R
@@ -30,6 +32,8 @@ class CommandR2Model(TextModel):
@ModelBase.register("Cohere2ForCausalLM")
# [TAG_HF_EXAMPLE_GATED] CohereLabs/c4ai-command-r7b-12-2024 is gated
@ModelBase.example("hf-tiny-v2/tiny-random-Cohere2ForCausalLM")
class Cohere2Model(TextModel):
model_arch = gguf.MODEL_ARCH.COHERE2
@@ -59,6 +63,7 @@ class Cohere2Model(TextModel):
@ModelBase.register("Cohere2MoeForCausalLM")
@ModelBase.example("CohereLabs/North-Mini-Code-1.0")
class Cohere2MoeModel(TextModel):
model_arch = gguf.MODEL_ARCH.COHERE2MOE
_n_main_layers: int | None = None
+1
View File
@@ -9,6 +9,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("DbrxForCausalLM")
@ModelBase.example("alpindale/dbrx-instruct")
class DbrxModel(TextModel):
model_arch = gguf.MODEL_ARCH.DBRX
+1
View File
@@ -13,6 +13,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("DeciLMForCausalLM")
@ModelBase.example("nvidia/Llama-3_1-Nemotron-51B-Instruct", "Deci/DeciLM-7B")
class DeciModel(TextModel):
model_arch = gguf.MODEL_ARCH.DECI
+8
View File
@@ -18,6 +18,7 @@ from .qwen import QwenModel
@ModelBase.register("DeepseekOCRForCausalLM")
@ModelBase.example("deepseek-ai/DeepSeek-OCR")
class DeepseekOCRVisionModel(MmprojModel):
# HF dynamic_preprocess() max_num, which differs per model
preproc_max_tiles = 9
@@ -100,11 +101,13 @@ class DeepseekOCRVisionModel(MmprojModel):
@ModelBase.register("UnlimitedOCRForCausalLM")
@ModelBase.example("baidu/Unlimited-OCR")
class UnlimitedOCRVisionModel(DeepseekOCRVisionModel):
preproc_max_tiles = 32
@ModelBase.register("DeepseekOCR2ForCausalLM")
@ModelBase.example("deepseek-ai/DeepSeek-OCR-2")
class DeepseekOCR2VisionModel(DeepseekOCRVisionModel):
preproc_max_tiles = 6
@@ -134,6 +137,7 @@ class DeepseekOCR2VisionModel(DeepseekOCRVisionModel):
@ModelBase.register("DeepseekForCausalLM")
@ModelBase.example("deepseek-ai/deepseek-moe-16b-chat")
class DeepseekModel(TextModel):
model_arch = gguf.MODEL_ARCH.DEEPSEEK
@@ -228,6 +232,7 @@ class DeepseekModel(TextModel):
"YoutuForCausalLM",
"YoutuVLForConditionalGeneration",
)
@ModelBase.example("deepseek-ai/DeepSeek-V2-Lite", "deepseek-ai/DeepSeek-V3")
class DeepseekV2Model(TextModel):
model_arch = gguf.MODEL_ARCH.DEEPSEEK2
@@ -457,6 +462,7 @@ class DeepseekV2Model(TextModel):
@ModelBase.register("DeepseekV32ForCausalLM")
@ModelBase.example("deepseek-ai/DeepSeek-V3.2-Exp")
class DeepseekV32Model(DeepseekV2Model):
model_arch = gguf.MODEL_ARCH.DEEPSEEK32
skip_mtp = False
@@ -517,6 +523,7 @@ class DeepseekV32Model(DeepseekV2Model):
@ModelBase.register("DeepseekV4ForCausalLM")
@ModelBase.example("deepseek-ai/DeepSeek-V4-Flash-Base")
class DeepseekV4Model(TextModel):
model_arch = gguf.MODEL_ARCH.DEEPSEEK4
supports_mtp_export = True
@@ -911,6 +918,7 @@ class DeepseekV4Model(TextModel):
@ModelBase.register("DeepseekV4DSparkModel")
@ModelBase.example("deepseek-ai/DeepSeek-V4-Flash-DSpark")
class DeepseekV4DSparkModel(DeepseekV4Model):
model_arch = gguf.MODEL_ARCH.DFLASH
+1
View File
@@ -11,6 +11,7 @@ from .qwen import Qwen2MoeModel
@ModelBase.register("Dots1ForCausalLM")
@ModelBase.example("rednote-hilab/dots.llm1.inst")
class Dots1Model(Qwen2MoeModel):
model_arch = gguf.MODEL_ARCH.DOTS1
+1
View File
@@ -9,6 +9,7 @@ from .base import MmprojModel, ModelBase, gguf
@ModelBase.register("DotsOCRForCausalLM")
@ModelBase.example("rednote-hilab/dots.ocr")
class DotsOCRVisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
+1
View File
@@ -9,6 +9,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("DreamModel")
@ModelBase.example("Dream-org/Dream-v0-Instruct-7B")
class DreamModel(TextModel):
model_arch = gguf.MODEL_ARCH.DREAM
+4
View File
@@ -15,6 +15,7 @@ from .base import MmprojModel, ModelBase, TextModel, gguf
@ModelBase.register("Ernie4_5_ForCausalLM", "Ernie4_5ForCausalLM")
@ModelBase.example("baidu/ERNIE-4.5-0.3B-PT")
class Ernie4_5Model(TextModel):
model_arch = gguf.MODEL_ARCH.ERNIE4_5
@@ -73,6 +74,7 @@ class Ernie4_5Model(TextModel):
@ModelBase.register("Ernie4_5_MoeForCausalLM")
@ModelBase.example("baidu/ERNIE-4.5-21B-A3B-PT")
class Ernie4_5MoeModel(Ernie4_5Model):
model_arch = gguf.MODEL_ARCH.ERNIE4_5_MOE
_experts: list[dict[str, Tensor]] | None = None
@@ -156,11 +158,13 @@ class Ernie4_5MoeModel(Ernie4_5Model):
@ModelBase.register("PaddleOCRVLForConditionalGeneration")
@ModelBase.example("PaddlePaddle/PaddleOCR-VL")
class PaddleOCRModel(Ernie4_5Model):
model_arch = gguf.MODEL_ARCH.PADDLEOCR
@ModelBase.register("PaddleOCRVisionModel")
@ModelBase.example("PaddlePaddle/PaddleOCR-VL")
class PaddleOCRVisionModel(MmprojModel):
# PaddleOCR-VL uses a modified version of Siglip
min_pixels: int = 0
+5
View File
@@ -15,6 +15,7 @@ from .qwenvl import Qwen2VLVisionModel
@ModelBase.register("ExaoneForCausalLM")
@ModelBase.example("LGAI-EXAONE/EXAONE-3.5-2.4B-Instruct")
class ExaoneModel(TextModel):
model_arch = gguf.MODEL_ARCH.EXAONE
@@ -60,6 +61,7 @@ class ExaoneModel(TextModel):
@ModelBase.register("Exaone4ForCausalLM")
@ModelBase.example("LGAI-EXAONE/EXAONE-4.0-32B")
class Exaone4Model(TextModel):
model_arch = gguf.MODEL_ARCH.EXAONE4
@@ -126,6 +128,7 @@ class Exaone4Model(TextModel):
# note: transformers >= 5.1 renamed the class to "ExaoneMoeForCausalLM" (lowercase 'e'),
# so accept both spellings - LG AI have updated the configs of already-released models
@ModelBase.register("ExaoneMoEForCausalLM", "ExaoneMoeForCausalLM")
@ModelBase.example("LGAI-EXAONE/K-EXAONE-236B-A23B")
class ExaoneMoEModel(Exaone4Model):
model_arch = gguf.MODEL_ARCH.EXAONE_MOE
@@ -214,6 +217,7 @@ class ExaoneMoEModel(Exaone4Model):
@ModelBase.register("Exaone4_5_ForConditionalGeneration")
@ModelBase.example("LGAI-EXAONE/EXAONE-4.5-33B")
class Exaone4_5_TextModel(Exaone4Model):
"""Text tower of EXAONE 4.5; Tensors match EXAONE4"""
@@ -267,6 +271,7 @@ class Exaone4_5_TextModel(Exaone4Model):
@ModelBase.register("Exaone4_5_ForConditionalGeneration")
@ModelBase.example("LGAI-EXAONE/EXAONE-4.5-33B")
class Exaone4_5VisionModel(Qwen2VLVisionModel):
"""Vision tower for EXAONE 4.5; Qwen2-VL-style ViT (GQA) + patch merger"""
+1
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("FalconForCausalLM", "RWForCausalLM")
@ModelBase.example("tiiuae/falcon-7b")
class FalconModel(TextModel):
model_arch = gguf.MODEL_ARCH.FALCON
+1
View File
@@ -12,6 +12,7 @@ from .mamba import Mamba2Model
@ModelBase.register("FalconH1ForCausalLM")
@ModelBase.example("tiiuae/Falcon-H1-0.5B-Base")
class FalconH1Model(Mamba2Model):
model_arch = gguf.MODEL_ARCH.FALCON_H1
+19
View File
@@ -14,6 +14,8 @@ from .base import MmprojModel, ModelBase, TextModel, gguf, logger
@ModelBase.register("GemmaForCausalLM")
# [TAG_HF_EXAMPLE_GATED] google/gemma-2b is gated
@ModelBase.example("trl-internal-testing/tiny-GemmaForCausalLM")
class GemmaModel(TextModel):
model_arch = gguf.MODEL_ARCH.GEMMA
@@ -68,6 +70,8 @@ class GemmaModel(TextModel):
@ModelBase.register("Gemma2ForCausalLM")
# [TAG_HF_EXAMPLE_GATED] google/gemma-2-9b-it is gated
@ModelBase.example("trl-internal-testing/tiny-Gemma2ForCausalLM")
class Gemma2Model(TextModel):
model_arch = gguf.MODEL_ARCH.GEMMA2
@@ -118,6 +122,8 @@ class Gemma2Model(TextModel):
@ModelBase.register("Gemma3ForCausalLM", "Gemma3ForConditionalGeneration")
# [TAG_HF_EXAMPLE_GATED] google/gemma-3-4b-it is gated
@ModelBase.example("trl-internal-testing/tiny-Gemma3ForConditionalGeneration", "hf-tiny-v2/tiny-random-Gemma3ForCausalLM")
class Gemma3Model(TextModel):
model_arch = gguf.MODEL_ARCH.GEMMA3
@@ -174,6 +180,8 @@ class Gemma3Model(TextModel):
@ModelBase.register("Gemma3TextModel")
# [TAG_HF_EXAMPLE_GATED] google/embeddinggemma-300m is gated
@ModelBase.example("hf-tiny-v2/tiny-random-Gemma3TextModel")
class EmbeddingGemma(Gemma3Model):
model_arch = gguf.MODEL_ARCH.GEMMA_EMBEDDING
module_paths = []
@@ -248,6 +256,8 @@ class EmbeddingGemma(Gemma3Model):
@ModelBase.register("Gemma3ForConditionalGeneration")
# [TAG_HF_EXAMPLE_GATED] google/gemma-3-4b-it is gated
@ModelBase.example("trl-internal-testing/tiny-Gemma3ForConditionalGeneration")
class Gemma3VisionModel(MmprojModel):
def set_gguf_parameters(self):
super().set_gguf_parameters()
@@ -352,6 +362,8 @@ class ConformerAudioModel(MmprojModel):
@ModelBase.register("Gemma3nForConditionalGeneration")
# [TAG_HF_EXAMPLE_GATED] google/gemma-3n-E2B-it is gated
@ModelBase.example("hf-tiny-v2/tiny-random-Gemma3nForConditionalGeneration")
class Gemma3nVisionAudioModel(ConformerAudioModel):
has_audio_encoder = True
has_vision_encoder = True
@@ -471,6 +483,8 @@ class Gemma3nVisionAudioModel(ConformerAudioModel):
@ModelBase.register("Gemma3nForCausalLM", "Gemma3nForConditionalGeneration")
# [TAG_HF_EXAMPLE_GATED] google/gemma-3n-E2B-it is gated
@ModelBase.example("hf-tiny-v2/tiny-random-Gemma3nForConditionalGeneration")
class Gemma3NModel(Gemma3Model):
model_arch = gguf.MODEL_ARCH.GEMMA3N
@@ -615,6 +629,7 @@ class Gemma3NModel(Gemma3Model):
@ModelBase.register("Gemma4ForConditionalGeneration", "Gemma4ForCausalLM")
@ModelBase.example("google/gemma-4-31B-it", "google/gemma-4-26B-A4B-it", "google/gemma-4-E2B-it")
class Gemma4Model(Gemma3Model):
model_arch = gguf.MODEL_ARCH.GEMMA4
@@ -795,6 +810,7 @@ class Gemma4Model(Gemma3Model):
@ModelBase.register("Gemma4UnifiedForConditionalGeneration")
@ModelBase.example("hf-tiny-v2/tiny-random-Gemma4UnifiedForConditionalGeneration")
class Gemma4UnifiedModel(Gemma4Model):
model_arch = gguf.MODEL_ARCH.GEMMA4
@@ -815,6 +831,7 @@ class Gemma4UnifiedModel(Gemma4Model):
@ModelBase.register("Gemma4AssistantForCausalLM", "Gemma4UnifiedAssistantForCausalLM")
@ModelBase.example("google/gemma-4-31B-it-assistant", "google/gemma-4-26B-A4B-it-assistant", "google/gemma-4-E2B-it-assistant")
class Gemma4AssistantModel(Gemma4Model):
model_arch = gguf.MODEL_ARCH.GEMMA4_ASSISTANT
@@ -835,6 +852,7 @@ class Gemma4AssistantModel(Gemma4Model):
@ModelBase.register("Gemma4ForConditionalGeneration")
@ModelBase.example("google/gemma-4-31B-it", "google/gemma-4-26B-A4B-it", "google/gemma-4-E2B-it")
class Gemma4VisionAudioModel(MmprojModel):
has_audio_encoder = True
has_vision_encoder = True
@@ -913,6 +931,7 @@ class Gemma4VisionAudioModel(MmprojModel):
@ModelBase.register("Gemma4UnifiedForConditionalGeneration")
@ModelBase.example("hf-tiny-v2/tiny-random-Gemma4UnifiedForConditionalGeneration")
class Gemma4UnifiedVisionAudioModel(Gemma4VisionAudioModel):
has_audio_encoder = True
has_vision_encoder = True
+6
View File
@@ -15,6 +15,7 @@ from .deepseek import DeepseekV2Model
@ModelBase.register("Glm4ForCausalLM", "Glm4vForConditionalGeneration")
@ModelBase.example("zai-org/GLM-4-9B-0414")
class Glm4Model(TextModel):
model_arch = gguf.MODEL_ARCH.GLM4
use_mrope = False
@@ -86,6 +87,7 @@ class Glm4Model(TextModel):
@ModelBase.register("GlmOcrForConditionalGeneration")
@ModelBase.example("zai-org/GLM-OCR")
class GlmOCRModel(Glm4Model):
model_arch = gguf.MODEL_ARCH.GLM4
use_mrope = False
@@ -107,6 +109,7 @@ class GlmOCRModel(Glm4Model):
@ModelBase.register("Glm4MoeForCausalLM", "Glm4vMoeForConditionalGeneration")
@ModelBase.example("zai-org/GLM-4.5-Air")
class Glm4MoeModel(TextModel):
model_arch = gguf.MODEL_ARCH.GLM4_MOE
@@ -204,6 +207,7 @@ class Glm4MoeModel(TextModel):
@ModelBase.register("Glm4MoeLiteForCausalLM")
@ModelBase.example("zai-org/GLM-4.7-Flash")
class Glm4MoeLiteModel(DeepseekV2Model):
model_arch = gguf.MODEL_ARCH.DEEPSEEK2
skip_mtp = False
@@ -272,6 +276,7 @@ class Glm4MoeLiteModel(DeepseekV2Model):
@ModelBase.register("GlmMoeDsaForCausalLM")
@ModelBase.example("zai-org/GLM-5.2")
class GlmMoeDsaModel(DeepseekV2Model):
model_arch = gguf.MODEL_ARCH.GLM_DSA
skip_mtp = False
@@ -340,6 +345,7 @@ class GlmMoeDsaModel(DeepseekV2Model):
@ModelBase.register("SolarOpenForCausalLM")
@ModelBase.example("upstage/Solar-Open-100B")
class SolarOpenModel(Glm4MoeModel):
model_arch = gguf.MODEL_ARCH.GLM4_MOE
+2
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("GPT2LMHeadModel")
@ModelBase.example("openai-community/gpt2")
class GPT2Model(TextModel):
model_arch = gguf.MODEL_ARCH.GPT2
@@ -38,6 +39,7 @@ class GPT2Model(TextModel):
@ModelBase.register("RuGPT3XLForCausalLM")
@ModelBase.example("evilfreelancer/ruGPT3XL")
class RuGPT3XLModel(TextModel):
model_arch = gguf.MODEL_ARCH.GPT2
+1
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("GptOssForCausalLM")
@ModelBase.example("openai/gpt-oss-20b")
class GptOssModel(TextModel):
model_arch = gguf.MODEL_ARCH.GPT_OSS
+1
View File
@@ -13,6 +13,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("GPTNeoXForCausalLM")
@ModelBase.example("EleutherAI/pythia-70m")
class GPTNeoXModel(TextModel):
model_arch = gguf.MODEL_ARCH.GPTNEOX
+7
View File
@@ -15,6 +15,7 @@ from .mamba import Mamba2Model
@ModelBase.register("GraniteForCausalLM")
@ModelBase.example("ibm-granite/granite-3.3-2b-instruct")
class GraniteModel(LlamaModel):
"""Conversion for IBM's GraniteForCausalLM"""
model_arch = gguf.MODEL_ARCH.GRANITE
@@ -74,6 +75,7 @@ class GraniteModel(LlamaModel):
@ModelBase.register("GraniteMoeForCausalLM", "GraniteMoeSharedForCausalLM")
@ModelBase.example("ibm-granite/granite-3.1-3b-a800m-instruct")
class GraniteMoeModel(GraniteModel):
"""Conversion for IBM's GraniteMoeForCausalLM"""
model_arch = gguf.MODEL_ARCH.GRANITE_MOE
@@ -124,6 +126,7 @@ class GraniteMoeModel(GraniteModel):
@ModelBase.register("GraniteSwitchForCausalLM")
@ModelBase.example("ibm-granite/granite-switch-4.1-3b-preview")
class GraniteSwitchModel(GraniteMoeModel):
"""Dense, all-attention Granite with N per-token embedded LoRA adapters, stacked
over the adapter dim with a zero adapter at slot 0 (N = num_adapters + 1)."""
@@ -284,6 +287,7 @@ class GraniteSwitchModel(GraniteMoeModel):
@ModelBase.register("GraniteMoeHybridForCausalLM", "BambaForCausalLM")
@ModelBase.example("ibm-granite/granite-4.0-h-tiny", "ibm-ai-platform/Bamba-9B-v2")
class GraniteHybridModel(Mamba2Model, GraniteMoeModel):
"""GraniteHybrid is a hybrid SSM + Attention model that uses Mamba2 SSM
layers and optionally uses MoE w/ a shared expert"""
@@ -426,6 +430,7 @@ class GraniteHybridModel(Mamba2Model, GraniteMoeModel):
@ModelBase.register("GraniteSpeechForConditionalGeneration")
@ModelBase.example("ibm-granite/granite-speech-3.3-2b", "ibm-granite/granite-4.0-1b-speech")
class GraniteSpeechMmprojModel(MmprojModel):
has_vision_encoder = False
has_audio_encoder = True
@@ -509,6 +514,7 @@ class GraniteSpeechMmprojModel(MmprojModel):
@ModelBase.register("GraniteSpeechPlusForConditionalGeneration")
@ModelBase.example("ibm-granite/granite-speech-4.1-2b-plus")
class GraniteSpeechPlusMmprojModel(GraniteSpeechMmprojModel):
"""Conversion for GraniteSpeechPlus - extends GraniteSpeech with feature layer concatenation"""
has_vision_encoder = False
@@ -537,6 +543,7 @@ class GraniteSpeechPlusMmprojModel(GraniteSpeechMmprojModel):
@ModelBase.register("Granite4VisionForConditionalGeneration")
@ModelBase.example("ibm-granite/granite-4.0-3b-vision")
class Granite4VisionMmprojModel(MmprojModel):
has_vision_encoder = True
has_audio_encoder = False
+1
View File
@@ -13,6 +13,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("GrokForCausalLM", "Grok1ForCausalLM")
@ModelBase.example("keyfan/grok-1-hf")
class GrokModel(TextModel):
model_arch = gguf.MODEL_ARCH.GROK
+1
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("GroveMoeForCausalLM", "modeling_grove_moe.GroveMoeForCausalLM")
@ModelBase.example("inclusionAI/GroveMoE-Inst")
class GroveMoeModel(TextModel):
model_arch = gguf.MODEL_ARCH.GROVEMOE
+5
View File
@@ -17,6 +17,7 @@ from .qwen import QwenModel
@ModelBase.register("HunYuanMoEV1ForCausalLM")
@ModelBase.example("tencent/Hunyuan-A13B-Instruct")
class HunYuanMoEModel(TextModel):
model_arch = gguf.MODEL_ARCH.HUNYUAN_MOE
@@ -154,6 +155,7 @@ class HunYuanMoEModel(TextModel):
@ModelBase.register("HunYuanDenseV1ForCausalLM")
@ModelBase.example("tencent/Hunyuan-4B-Instruct")
class HunYuanModel(TextModel):
model_arch = gguf.MODEL_ARCH.HUNYUAN_DENSE
@@ -290,6 +292,7 @@ class HunYuanModel(TextModel):
@ModelBase.register("HunYuanVLForConditionalGeneration")
@ModelBase.example("tencent/HunyuanOCR")
class HunyuanVLVisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
@@ -333,6 +336,7 @@ class HunyuanVLVisionModel(MmprojModel):
@ModelBase.register("HunYuanVLForConditionalGeneration")
@ModelBase.example("tencent/HunyuanOCR")
class HunyuanVLTextModel(HunYuanModel):
model_arch = gguf.MODEL_ARCH.HUNYUAN_VL
@@ -365,6 +369,7 @@ class HunyuanVLTextModel(HunYuanModel):
@ModelBase.register("HYV3ForCausalLM")
@ModelBase.example("tencent/Hy3")
class HYV3Model(TextModel):
model_arch = gguf.MODEL_ARCH.HY_V3
supports_mtp_export = True
+2
View File
@@ -14,6 +14,7 @@ from .llama import LlamaModel
@ModelBase.register("InternLM2ForCausalLM")
@ModelBase.example("internlm/internlm2-chat-7b")
class InternLM2Model(TextModel):
model_arch = gguf.MODEL_ARCH.INTERNLM2
@@ -170,6 +171,7 @@ class InternLM2Model(TextModel):
@ModelBase.register("InternLM3ForCausalLM")
@ModelBase.example("internlm/internlm3-8b-instruct")
class InternLM3Model(TextModel):
model_arch = gguf.MODEL_ARCH.LLAMA
+1
View File
@@ -9,6 +9,7 @@ from .base import MmprojModel, ModelBase, gguf
@ModelBase.register("InternVisionModel")
@ModelBase.example("OpenGVLab/InternVL3-2B", "OpenGVLab/InternVL2_5-1B")
class InternVisionModel(MmprojModel):
min_dynamic_tiles: int = 0
+3
View File
@@ -11,6 +11,8 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("Jais2ForCausalLM")
# [TAG_HF_EXAMPLE_GATED] inceptionai/Jais-2-8B-Chat is gated
# [TAG_HF_EXAMPLE_MISSING]
class Jais2Model(TextModel):
model_arch = gguf.MODEL_ARCH.JAIS2
@@ -22,6 +24,7 @@ class Jais2Model(TextModel):
@ModelBase.register("JAISLMHeadModel")
@ModelBase.example("inceptionai/jais-family-590m")
class JaisModel(TextModel):
model_arch = gguf.MODEL_ARCH.JAIS
+1
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("JambaForCausalLM")
@ModelBase.example("ai21labs/Jamba-v0.1")
class JambaModel(TextModel):
model_arch = gguf.MODEL_ARCH.JAMBA
+2
View File
@@ -11,6 +11,7 @@ from .llama import LlamaModel
@ModelBase.register("JanusForConditionalGeneration")
@ModelBase.example("deepseek-community/Janus-Pro-1B")
class JanusProModel(LlamaModel):
model_arch = gguf.MODEL_ARCH.LLAMA # reuse Llama arch
@@ -34,6 +35,7 @@ class JanusProModel(LlamaModel):
@ModelBase.register("JanusForConditionalGeneration")
@ModelBase.example("deepseek-community/Janus-Pro-1B")
class JanusProVisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
+1
View File
@@ -16,6 +16,7 @@ from .kimi_linear import KimiLinearModel
@ModelBase.register("KimiK3ForConditionalGeneration")
@ModelBase.example("moonshotai/Kimi-K3")
class KimiK3Model(TextModel):
"""
Kimi-K3 text model (KimiLinearForCausalLM under a `language_model.` prefix).
+1
View File
@@ -13,6 +13,7 @@ from .qwen import QwenModel
@ModelBase.register("KimiLinearModel", "KimiLinearForCausalLM")
@ModelBase.example("moonshotai/Kimi-Linear-48B-A3B-Instruct")
class KimiLinearModel(TextModel):
"""Kimi-Linear model with hybrid MLA+KDA architecture"""
model_arch = gguf.MODEL_ARCH.KIMI_LINEAR
+3
View File
@@ -11,6 +11,7 @@ from .base import MmprojModel, ModelBase, gguf
@ModelBase.register("KimiVLForConditionalGeneration")
@ModelBase.example("moonshotai/Kimi-VL-A3B-Instruct")
class KimiVLModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
@@ -52,6 +53,7 @@ class KimiVLModel(MmprojModel):
@ModelBase.register("KimiK25ForConditionalGeneration")
@ModelBase.example("moonshotai/Kimi-K2.5")
class KimiK25Model(MmprojModel):
"""Kimi-K2.5 with MoonViT3d vision encoder"""
@@ -155,6 +157,7 @@ class KimiK25Model(MmprojModel):
@ModelBase.register("Glm5vForConditionalGeneration")
# [TAG_HF_EXAMPLE_MISSING]
class Glm5vModel(KimiK25Model):
"""GLM-5.2-Vision MoonViT3d encoder and projector
+1
View File
@@ -13,6 +13,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("LagunaForCausalLM")
@ModelBase.example("poolside/Laguna-XS.2", "poolside/Laguna-S-2.1")
class LagunaModel(TextModel):
model_arch = gguf.MODEL_ARCH.LAGUNA
_experts: list[dict] | None = None
+6
View File
@@ -13,6 +13,7 @@ from .gemma import ConformerAudioModel
@ModelBase.register("Lfm2ForCausalLM", "LFM2ForCausalLM")
@ModelBase.example("LiquidAI/LFM2-1.2B", "LiquidAI/LFM2.5-350M")
class LFM2Model(TextModel):
model_arch = gguf.MODEL_ARCH.LFM2
@@ -65,6 +66,7 @@ class LFM2Model(TextModel):
@ModelBase.register("Lfm2Model", "Lfm2BidirectionalModel")
@ModelBase.example("LiquidAI/LFM2.5-ColBERT-350M", "LiquidAI/LFM2.5-Embedding-350M")
class LFM2ColBertModel(LFM2Model):
model_arch = gguf.MODEL_ARCH.LFM2
dense_tensor_name = "dense_2"
@@ -93,6 +95,7 @@ class LFM2ColBertModel(LFM2Model):
@ModelBase.register("Lfm2MoeForCausalLM")
@ModelBase.example("LiquidAI/LFM2-8B-A1B")
class LFM2MoeModel(TextModel):
model_arch = gguf.MODEL_ARCH.LFM2MOE
@@ -166,6 +169,7 @@ class LFM2MoeModel(TextModel):
@ModelBase.register("Lfm2VlForConditionalGeneration")
@ModelBase.example("LiquidAI/LFM2-VL-450M")
class LFM2VLModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
@@ -200,6 +204,7 @@ class LFM2VLModel(MmprojModel):
@ModelBase.register("Lfm2AudioForConditionalGeneration")
@ModelBase.example("LiquidAI/LFM2.5-Audio-1.5B", "LiquidAI/LFM2-Audio-1.5B")
class LFM2AudioModel(ConformerAudioModel):
has_vision_encoder = False
has_audio_encoder = True
@@ -238,6 +243,7 @@ class LFM2AudioModel(ConformerAudioModel):
@ModelBase.register("Lfm25AudioTokenizer")
@ModelBase.example("LiquidAI/LFM2.5-Audio-1.5B")
class LFM25AudioTokenizer(LFM2Model):
model_arch = gguf.MODEL_ARCH.LFM2
+1
View File
@@ -11,6 +11,7 @@ from .llava import LlavaVisionModel
@ModelBase.register("LightOnOCRForConditionalGeneration")
@ModelBase.example("lightonai/LightOnOCR-1B-1025")
class LightOnOCRVisionModel(LlavaVisionModel):
is_mistral_format = False
use_break_tok = False
+2
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("LLaDAModelLM")
@ModelBase.example("GSAI-ML/LLaDA-8B-Instruct")
class LLaDAModel(TextModel):
model_arch = gguf.MODEL_ARCH.LLADA
undo_permute = True
@@ -114,6 +115,7 @@ class LLaDAModel(TextModel):
@ModelBase.register("LLaDAMoEModel", "LLaDAMoEModelLM")
@ModelBase.example("inclusionAI/LLaDA-MoE-7B-A1B-Instruct")
class LLaDAMoEModel(TextModel):
model_arch = gguf.MODEL_ARCH.LLADA_MOE
+8
View File
@@ -28,6 +28,8 @@ from .base import ModelBase, TextModel, gguf, logger
"Eagle3DraftModel",
"IQuestCoderForCausalLM",
"LlamaModel")
# [TAG_HF_EXAMPLE_GATED] meta-llama/Llama-3.2-1B-Instruct is gated
@ModelBase.example("unsloth/Llama-3.2-1B-Instruct", "mistralai/Mistral-7B-Instruct-v0.3", "mistralai/Mixtral-8x7B-Instruct-v0.1")
class LlamaModel(TextModel):
model_arch = gguf.MODEL_ARCH.LLAMA
undo_permute = True
@@ -359,6 +361,7 @@ class LlamaModel(TextModel):
@ModelBase.register("ArceeForCausalLM")
@ModelBase.example("arcee-ai/AFM-4.5B")
class ArceeModel(LlamaModel):
model_arch = gguf.MODEL_ARCH.ARCEE
@@ -371,6 +374,8 @@ class ArceeModel(LlamaModel):
"Llama4ForConditionalGeneration",
"Llama4ForCausalLM",
)
# [TAG_HF_EXAMPLE_GATED] meta-llama/Llama-4-Scout-17B-16E-Instruct is gated
@ModelBase.example("unsloth/Llama-4-Scout-17B-16E-Instruct")
class Llama4Model(LlamaModel):
model_arch = gguf.MODEL_ARCH.LLAMA4
undo_permute = False
@@ -412,16 +417,19 @@ class Llama4Model(LlamaModel):
@ModelBase.register("LlamaBidirectionalModel")
@ModelBase.example("nvidia/llama-embed-nemotron-8b")
class LlamaEmbedNemotronModel(LlamaModel):
model_arch = gguf.MODEL_ARCH.LLAMA_EMBED
@ModelBase.register("SmolLM3ForCausalLM")
@ModelBase.example("HuggingFaceTB/SmolLM3-3B")
class SmolLM3Model(LlamaModel):
model_arch = gguf.MODEL_ARCH.SMOLLM3
@ModelBase.register("ApertusForCausalLM")
@ModelBase.example("swiss-ai/Apertus-8B-Instruct-2509")
class ApertusModel(LlamaModel):
model_arch = gguf.MODEL_ARCH.APERTUS
undo_permute = False
+2
View File
@@ -9,6 +9,8 @@ from .base import MmprojModel, ModelBase, gguf
@ModelBase.register("Llama4ForConditionalGeneration")
# [TAG_HF_EXAMPLE_GATED] meta-llama/Llama-4-Scout-17B-16E-Instruct is gated
@ModelBase.example("unsloth/Llama-4-Scout-17B-16E-Instruct")
class Llama4VisionModel(MmprojModel):
def set_gguf_parameters(self):
super().set_gguf_parameters()
+1
View File
@@ -16,6 +16,7 @@ from .llama import LlamaModel
"LlavaForConditionalGeneration", # pixtral
"Mistral3ForConditionalGeneration", # mistral small 3.1
)
@ModelBase.example("mistral-community/pixtral-12b", "mistralai/Mistral-Small-3.1-24B-Instruct-2503")
class LlavaVisionModel(MmprojModel):
img_break_tok_id = -1
use_break_tok = True
+1
View File
@@ -4,6 +4,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("MaincoderForCausalLM")
@ModelBase.example("Maincode/Maincoder-1B")
class MaincoderModel(TextModel):
model_arch = gguf.MODEL_ARCH.MAINCODER
+2
View File
@@ -14,6 +14,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("MambaForCausalLM", "MambaLMHeadModel", "FalconMambaForCausalLM")
@ModelBase.example("state-spaces/mamba-130m-hf", "tiiuae/falcon-mamba-7b")
class MambaModel(TextModel):
model_arch = gguf.MODEL_ARCH.MAMBA
@@ -100,6 +101,7 @@ class MambaModel(TextModel):
@ModelBase.register("Mamba2ForCausalLM")
@ModelBase.example("mistralai/Mamba-Codestral-7B-v0.1")
class Mamba2Model(TextModel):
model_arch = gguf.MODEL_ARCH.MAMBA2
+1
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("MellumForCausalLM")
@ModelBase.example("JetBrains/Mellum2-12B-A2.5B-Base")
class MellumModel(TextModel):
model_arch = gguf.MODEL_ARCH.MELLUM
+2
View File
@@ -14,6 +14,7 @@ from .base import MmprojModel, ModelBase, TextModel, gguf
@ModelBase.register("MiMoV2FlashForCausalLM", "MiMoV2ForCausalLM")
@ModelBase.example("XiaomiMiMo/MiMo-V2.5")
class MimoV2Model(TextModel):
model_arch = gguf.MODEL_ARCH.MIMO2
@@ -230,6 +231,7 @@ class MimoV2Model(TextModel):
@ModelBase.register("MiMoV2ForCausalLM")
@ModelBase.example("XiaomiMiMo/MiMo-V2.5")
class MiMoV2VisionAudioModel(MmprojModel):
has_audio_encoder = True
+4
View File
@@ -14,6 +14,7 @@ from .qwen import Qwen3_5TextModel
@ModelBase.register("MiniCPMForCausalLM")
@ModelBase.example("openbmb/MiniCPM-2B-sft-bf16")
class MiniCPMModel(TextModel):
model_arch = gguf.MODEL_ARCH.MINICPM
@@ -61,6 +62,7 @@ class MiniCPMModel(TextModel):
@ModelBase.register("MiniCPM3ForCausalLM")
@ModelBase.example("openbmb/MiniCPM3-4B")
class MiniCPM3Model(TextModel):
model_arch = gguf.MODEL_ARCH.MINICPM3
@@ -117,6 +119,7 @@ class MiniCPM3Model(TextModel):
# the LM (text mode) and once as the mmproj (vision mode), mirroring the Qwen3-VL setup.
@ModelBase.register("MiniCPMV4_6ForConditionalGeneration")
@ModelBase.example("openbmb/MiniCPM-V-4_6")
class MiniCPMV4_6TextModel(Qwen3_5TextModel):
model_arch = gguf.MODEL_ARCH.QWEN35
@@ -134,6 +137,7 @@ class MiniCPMV4_6TextModel(Qwen3_5TextModel):
@ModelBase.register("MiniCPMV4_6ForConditionalGeneration")
@ModelBase.example("openbmb/MiniCPM-V-4_6")
class MiniCPMV4_6VisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
+4
View File
@@ -12,6 +12,7 @@ from .base import ModelBase, TextModel, MmprojModel, gguf, logger
@ModelBase.register("MiniMaxText01ForCausalLM")
@ModelBase.register("MiniMaxM1ForCausalLM")
@ModelBase.example("MiniMaxAI/MiniMax-Text-01", "MiniMaxAI/MiniMax-M1-40k")
class MiniMaxText01Model(TextModel):
model_arch = gguf.MODEL_ARCH.MINIMAX01
@@ -119,6 +120,7 @@ class MiniMaxText01Model(TextModel):
@ModelBase.register("MiniMaxM2ForCausalLM")
@ModelBase.example("MiniMaxAI/MiniMax-M2")
class MiniMaxM2Model(TextModel):
model_arch = gguf.MODEL_ARCH.MINIMAXM2
_experts_cache: dict[int, dict[str, Tensor]] = {}
@@ -163,6 +165,7 @@ class MiniMaxM2Model(TextModel):
@ModelBase.register("MiniMaxM3SparseForCausalLM", "MiniMaxM3SparseForConditionalGeneration")
@ModelBase.example("MiniMaxAI/MiniMax-M3")
class MiniMaxM3Model(MiniMaxM2Model):
model_arch = gguf.MODEL_ARCH.MINIMAXM3
@@ -203,6 +206,7 @@ class MiniMaxM3Model(MiniMaxM2Model):
@ModelBase.register("MiniMaxM3SparseForConditionalGeneration", "MiniMaxM3VLForConditionalGeneration")
@ModelBase.example("MiniMaxAI/MiniMax-M3")
class MiniMaxM3VisionModel(MmprojModel):
@classmethod
def filter_tensors(cls, item):
+1
View File
@@ -15,6 +15,7 @@ from .llama import LlamaModel
"Mistral3ForConditionalGeneration",
"Ministral3ForCausalLM",
)
@ModelBase.example("mistralai/Mistral-Small-3.1-24B-Instruct-2503", "hf-tiny-v2/tiny-random-Ministral3ForCausalLM")
class Mistral3Model(TextModel):
class Ministral3Model(LlamaModel):
model_arch = gguf.MODEL_ARCH.MISTRAL3
+1
View File
@@ -9,6 +9,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("MPTForCausalLM")
@ModelBase.example("anas-awadalla/mpt-7b")
class MPTModel(TextModel):
model_arch = gguf.MODEL_ARCH.MPT
+3
View File
@@ -24,6 +24,7 @@ def _unpermute_for_rope(tensor: "Tensor", n_heads: int) -> "Tensor":
@ModelBase.register("MuseGlimmerForConditionalGeneration")
@ModelBase.example("meta-models/Muse-Glimmer-30B")
class MuseGlimmerModel(TextModel):
model_arch = gguf.MODEL_ARCH.MUSE_GLIMMER
@@ -78,6 +79,7 @@ class MuseGlimmerModel(TextModel):
@ModelBase.register("MuseGlimmerForConditionalGeneration")
@ModelBase.example("meta-models/Muse-Glimmer-30B")
class MuseGlimmerVisionModel(MmprojModel):
def get_vision_config(self) -> dict[str, Any] | None:
c = self.global_config.get("vision_config")
@@ -131,6 +133,7 @@ class MuseGlimmerVisionModel(MmprojModel):
@ModelBase.register("MuseGlimmerAssistantModel")
@ModelBase.example("meta-models/Muse-Glimmer-30B-assistant")
class MuseGlimmerAssistantModel(TextModel):
model_arch = gguf.MODEL_ARCH.DFLASH
+1
View File
@@ -5,6 +5,7 @@ from .llama import LlamaModel
@ModelBase.register("NanbeigeForCausalLM")
@ModelBase.example("Nanbeige/Nanbeige4.2-3B")
class NanbeigeModel(LlamaModel):
model_arch = gguf.MODEL_ARCH.NANBEIGE
undo_permute = True
+3
View File
@@ -16,6 +16,7 @@ from .granite import GraniteHybridModel
"NemotronH_Nano_VL_V2",
"RADIOModel",
)
@ModelBase.example("nvidia/NVIDIA-Nemotron-Nano-12B-v2-VL-BF16")
class NemotronNanoV2VLModel(MmprojModel):
# ViT-Huge architecture parameters for RADIO v2.5-h
_vit_hidden_size = 1280
@@ -151,6 +152,7 @@ class NemotronNanoV2VLModel(MmprojModel):
@ModelBase.register("NemotronForCausalLM")
@ModelBase.example("nvidia/Minitron-4B-Base")
class NemotronModel(TextModel):
model_arch = gguf.MODEL_ARCH.NEMOTRON
@@ -193,6 +195,7 @@ class NemotronModel(TextModel):
@ModelBase.register("NemotronHForCausalLM")
@ModelBase.example("nvidia/Nemotron-H-8B-Base-8K")
class NemotronHModel(GraniteHybridModel):
"""Hybrid mamba2/attention model from NVIDIA"""
model_arch = gguf.MODEL_ARCH.NEMOTRON_H
+4
View File
@@ -14,6 +14,7 @@ from .llama import LlamaModel
@ModelBase.register("OlmoForCausalLM")
@ModelBase.register("OLMoForCausalLM")
@ModelBase.example("allenai/OLMo-1.7-7B-hf")
class OlmoModel(TextModel):
model_arch = gguf.MODEL_ARCH.OLMO
@@ -39,12 +40,14 @@ class OlmoModel(TextModel):
@ModelBase.register("SeedOssForCausalLM")
@ModelBase.example("ByteDance-Seed/Seed-OSS-36B-Instruct")
class SeedOssModel(TextModel):
model_arch = gguf.MODEL_ARCH.SEED_OSS
@ModelBase.register("Olmo2ForCausalLM")
@ModelBase.register("Olmo3ForCausalLM")
@ModelBase.example("allenai/OLMo-2-1124-7B-Instruct", "allenai/Olmo-3-7B-Instruct")
class Olmo2Model(TextModel):
model_arch = gguf.MODEL_ARCH.OLMO2
@@ -67,6 +70,7 @@ class Olmo2Model(TextModel):
@ModelBase.register("OlmoeForCausalLM")
@ModelBase.example("allenai/OLMoE-1B-7B-0924")
class OlmoeModel(TextModel):
model_arch = gguf.MODEL_ARCH.OLMOE
+1
View File
@@ -9,6 +9,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("OpenELMForCausalLM")
@ModelBase.example("apple/OpenELM-270M")
class OpenELMModel(TextModel):
model_arch = gguf.MODEL_ARCH.OPENELM
+1
View File
@@ -4,6 +4,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("OrionForCausalLM")
@ModelBase.example("OrionStarAI/Orion-14B-Base")
class OrionModel(TextModel):
model_arch = gguf.MODEL_ARCH.ORION
+1
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("PanguEmbeddedForCausalLM")
@ModelBase.example("FreedomIntelligence/openPangu-Embedded-7B-V1.1")
class PanguEmbeddedModel(TextModel):
model_arch = gguf.MODEL_ARCH.PANGU_EMBED
+4
View File
@@ -14,6 +14,7 @@ from .base import MmprojModel, ModelBase, SentencePieceTokenTypes, TextModel, gg
@ModelBase.register("PhiForCausalLM")
@ModelBase.example("microsoft/phi-2")
class Phi2Model(TextModel):
model_arch = gguf.MODEL_ARCH.PHI2
@@ -36,6 +37,7 @@ class Phi2Model(TextModel):
@ModelBase.register("Phi3ForCausalLM", "Phi4ForCausalLMV")
@ModelBase.example("microsoft/Phi-3-mini-4k-instruct")
class Phi3MiniModel(TextModel):
model_arch = gguf.MODEL_ARCH.PHI3
@@ -210,6 +212,7 @@ class Phi3MiniModel(TextModel):
@ModelBase.register("Phi4ForCausalLMV")
# [TAG_HF_EXAMPLE_MISSING]
class Phi4VisionMmprojModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
@@ -336,6 +339,7 @@ class Phi4VisionMmprojModel(MmprojModel):
@ModelBase.register("PhiMoEForCausalLM")
@ModelBase.example("microsoft/Phi-3.5-MoE-instruct")
class PhiMoeModel(Phi3MiniModel):
model_arch = gguf.MODEL_ARCH.PHIMOE
+4
View File
@@ -13,6 +13,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("PlamoForCausalLM")
@ModelBase.example("pfnet/plamo-13b")
class PlamoModel(TextModel):
model_arch = gguf.MODEL_ARCH.PLAMO
@@ -58,6 +59,7 @@ class PlamoModel(TextModel):
@ModelBase.register("Plamo2ForCausalLM", "PLaMo2ForCausalLM")
@ModelBase.example("pfnet/plamo-2-1b")
class Plamo2Model(TextModel):
model_arch = gguf.MODEL_ARCH.PLAMO2
@@ -147,6 +149,8 @@ class Plamo2Model(TextModel):
@ModelBase.register("Plamo3ForCausalLM", "PLaMo3ForCausalLM")
# [TAG_HF_EXAMPLE_GATED] pfnet/plamo-3-nict-2b-base is gated
@ModelBase.example("midorin-Linux/plamo-3-12b-self-merged-base")
class Plamo3Model(TextModel):
model_arch = gguf.MODEL_ARCH.PLAMO3
+1
View File
@@ -4,6 +4,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("PLMForCausalLM")
@ModelBase.example("PLM-Team/PLM-1.8B-Instruct")
class PLMModel(TextModel):
model_arch = gguf.MODEL_ARCH.PLM
+2
View File
@@ -77,6 +77,7 @@ def _load_hparams(dir_model: Path) -> dict[str, Any]:
@ModelBase.register("PocketTTSModel")
# [TAG_HF_EXAMPLE_MISSING] model is gated, and the checkpoint requires cd to subdir, not supported here
class PocketTTSModel(TextModel):
model_arch = gguf.MODEL_ARCH.POCKETTTS
@@ -174,6 +175,7 @@ class PocketTTSModel(TextModel):
@ModelBase.register("PocketTTSModel")
# [TAG_HF_EXAMPLE_MISSING] model is gated, and the checkpoint requires cd to subdir, not supported here
class PocketTTSMmprojModel(MmprojModel):
has_audio_encoder = True
has_vision_encoder = False
+11
View File
@@ -13,6 +13,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("QWenLMHeadModel")
@ModelBase.example("Qwen/Qwen-7B")
class QwenModel(TextModel):
model_arch = gguf.MODEL_ARCH.QWEN
@@ -51,6 +52,7 @@ class QwenModel(TextModel):
"AudioFlamingo3ForConditionalGeneration",
"DotsOCRForCausalLM",
)
@ModelBase.example("Qwen/Qwen2.5-7B-Instruct")
class Qwen2Model(TextModel):
model_arch = gguf.MODEL_ARCH.QWEN2
@@ -71,6 +73,7 @@ class Qwen2Model(TextModel):
@ModelBase.register("Qwen2MoeForCausalLM")
@ModelBase.example("Qwen/Qwen1.5-MoE-A2.7B")
class Qwen2MoeModel(TextModel):
model_arch = gguf.MODEL_ARCH.QWEN2MOE
@@ -153,6 +156,7 @@ class Qwen2MoeModel(TextModel):
@ModelBase.register("Qwen3ForCausalLM", "Qwen3Model")
@ModelBase.example("Qwen/Qwen3-8B")
class Qwen3Model(Qwen2Model):
model_arch = gguf.MODEL_ARCH.QWEN3
@@ -251,6 +255,7 @@ class Qwen3Model(Qwen2Model):
@ModelBase.register("Qwen3MoeForCausalLM")
@ModelBase.example("Qwen/Qwen3-30B-A3B")
class Qwen3MoeModel(Qwen2MoeModel):
model_arch = gguf.MODEL_ARCH.QWEN3MOE
@@ -362,6 +367,7 @@ class _QwenMtpMixin:
@ModelBase.register("Qwen3NextForCausalLM")
@ModelBase.example("Qwen/Qwen3-Next-80B-A3B-Instruct")
class Qwen3NextModel(_QwenMtpMixin, Qwen2MoeModel):
model_arch = gguf.MODEL_ARCH.QWEN3NEXT
@@ -421,6 +427,7 @@ class Qwen3NextModel(_QwenMtpMixin, Qwen2MoeModel):
@ModelBase.register("RND1")
@ModelBase.example("radicalnumerics/RND1-Base-0910")
class RND1Model(Qwen2MoeModel):
model_arch = gguf.MODEL_ARCH.RND1
@@ -620,16 +627,19 @@ class _Qwen35MRopeMixin:
@ModelBase.register("Qwen3_5ForConditionalGeneration", "Qwen3_5ForCausalLM")
@ModelBase.example("Qwen/Qwen3.5-9B")
class Qwen3_5TextModel(_Qwen35MRopeMixin, _LinearAttentionVReorderBase):
model_arch = gguf.MODEL_ARCH.QWEN35
@ModelBase.register("Qwen3_5MoeForConditionalGeneration", "Qwen3_5MoeForCausalLM")
@ModelBase.example("Qwen/Qwen3.5-35B-A3B")
class Qwen3_5MoeTextModel(_Qwen35MRopeMixin, _LinearAttentionVReorderBase):
model_arch = gguf.MODEL_ARCH.QWEN35MOE
@ModelBase.register("DFlashDraftModel")
@ModelBase.example("z-lab/Qwen3.5-9B-DFlash")
class DFlashModel(Qwen3Model):
model_arch = gguf.MODEL_ARCH.DFLASH
@@ -699,6 +709,7 @@ class DFlashModel(Qwen3Model):
@ModelBase.register("Qwen3DSparkModel")
@ModelBase.example("satgeze/Qwen3.6-27B-DSpark")
class DSparkModel(DFlashModel):
# DSpark = DFlash + a semi-autoregressive Markov head
model_arch = gguf.MODEL_ARCH.DFLASH
+2
View File
@@ -37,6 +37,7 @@ _ACT2FN = {
@ModelBase.register("Qwen3TTSForConditionalGeneration")
@ModelBase.example("Qwen/Qwen3-TTS-12Hz-1.7B-Base")
class Qwen3TTSTalkerModel(TextModel):
model_arch = gguf.MODEL_ARCH.QWEN3TTS
@@ -185,6 +186,7 @@ class Qwen3TTSTalkerModel(TextModel):
@ModelBase.register("Qwen3TTSForConditionalGeneration")
@ModelBase.example("Qwen/Qwen3-TTS-12Hz-1.7B-Base")
class Qwen3TTSSpeakerEncoderModel(MmprojModel):
has_vision_encoder = False
has_audio_encoder = True
+8
View File
@@ -14,6 +14,7 @@ from .qwenvl import Qwen25AudioModel
@ModelBase.register("Qwen3VLForConditionalGeneration", "Qwen3VLMoeForConditionalGeneration", "Qwen3_5ForConditionalGeneration", "Qwen3_5MoeForConditionalGeneration")
@ModelBase.example("Qwen/Qwen3-VL-4B-Instruct", "Qwen/Qwen3-VL-30B-A3B-Instruct", "Qwen/Qwen3.5-9B", "Qwen/Qwen3.5-35B-A3B")
class Qwen3VLVisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
@@ -144,6 +145,7 @@ class Qwen3VLVisionModel(MmprojModel):
@ModelBase.register("Qwen3OmniMoeForConditionalGeneration")
@ModelBase.example("Qwen/Qwen3-Omni-30B-A3B-Instruct")
class Qwen3OmniMmprojModel(Qwen3VLVisionModel, Qwen25AudioModel):
has_audio_encoder = True
has_vision_encoder = True
@@ -217,12 +219,14 @@ class Qwen3OmniMmprojModel(Qwen3VLVisionModel, Qwen25AudioModel):
@ModelBase.register("Qwen3ASRForConditionalGeneration")
@ModelBase.example("Qwen/Qwen3-ASR-0.6B-hf")
class Qwen3ASRMmprojModel(Qwen3OmniMmprojModel):
has_audio_encoder = True
has_vision_encoder = False
@ModelBase.register("Glm4vForConditionalGeneration", "Glm4vMoeForConditionalGeneration", "GlmOcrForConditionalGeneration")
@ModelBase.example("zai-org/GLM-4.1V-9B-Thinking", "zai-org/GLM-4.5V")
class Glm4VVisionModel(Qwen3VLVisionModel):
def set_gguf_parameters(self):
MmprojModel.set_gguf_parameters(self) # skip Qwen3VLVisionModel parameters
@@ -246,6 +250,7 @@ class Glm4VVisionModel(Qwen3VLVisionModel):
@ModelBase.register("Qwen3VLForConditionalGeneration")
@ModelBase.example("Qwen/Qwen3-VL-4B-Instruct")
class Qwen3VLTextModel(Qwen3Model):
model_arch = gguf.MODEL_ARCH.QWEN3VL
@@ -268,6 +273,7 @@ class Qwen3VLTextModel(Qwen3Model):
@ModelBase.register("Qwen3VLMoeForConditionalGeneration")
@ModelBase.example("Qwen/Qwen3-VL-30B-A3B-Instruct")
class Qwen3VLMoeTextModel(Qwen3MoeModel):
model_arch = gguf.MODEL_ARCH.QWEN3VLMOE
@@ -317,6 +323,7 @@ class Qwen3VLMoeTextModel(Qwen3MoeModel):
@ModelBase.register("Qwen3OmniMoeForConditionalGeneration")
@ModelBase.example("Qwen/Qwen3-Omni-30B-A3B-Instruct")
class Qwen3OmniMoeTextModel(Qwen3VLMoeTextModel):
model_arch = gguf.MODEL_ARCH.QWEN3VLMOE
@@ -338,6 +345,7 @@ class Qwen3OmniMoeTextModel(Qwen3VLMoeTextModel):
@ModelBase.register("Qwen3ASRForConditionalGeneration")
@ModelBase.example("Qwen/Qwen3-ASR-0.6B-hf")
class Qwen3ASRTextModel(Qwen3VLTextModel):
model_arch = gguf.MODEL_ARCH.QWEN3VL
+3
View File
@@ -17,6 +17,7 @@ from .base import MmprojModel, ModelBase, TextModel, gguf
"Qwen2_5_VLForConditionalGeneration",
"Qwen2_5OmniModel",
)
@ModelBase.example("Qwen/Qwen2-VL-2B-Instruct", "Qwen/Qwen2.5-VL-3B-Instruct")
class Qwen2VLModel(TextModel):
model_arch = gguf.MODEL_ARCH.QWEN2VL
@@ -40,6 +41,7 @@ class Qwen2VLModel(TextModel):
@ModelBase.register("Qwen2VLModel", "Qwen2VLForConditionalGeneration", "Qwen2_5_VLForConditionalGeneration")
@ModelBase.example("Qwen/Qwen2-VL-2B-Instruct", "Qwen/Qwen2.5-VL-3B-Instruct")
class Qwen2VLVisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
@@ -161,6 +163,7 @@ class Qwen25AudioModel(MmprojModel):
@ModelBase.register("Qwen2_5OmniModel")
@ModelBase.example("Qwen/Qwen2.5-Omni-3B")
class Qwen25OmniModel(Qwen2VLVisionModel, Qwen25AudioModel):
has_audio_encoder = True
has_vision_encoder = True
+1
View File
@@ -9,6 +9,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("GPTRefactForCausalLM")
@ModelBase.example("smallcloudai/Refact-1_6-base")
class RefactModel(TextModel):
model_arch = gguf.MODEL_ARCH.REFACT
+4
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("Rwkv6ForCausalLM")
@ModelBase.example("RWKV/v6-Finch-1B6-HF")
class Rwkv6Model(TextModel):
model_arch = gguf.MODEL_ARCH.RWKV6
@@ -83,6 +84,7 @@ class Rwkv6Model(TextModel):
@ModelBase.register("RWKV6Qwen2ForCausalLM")
@ModelBase.example("recursal/QRWKV6-32B-Instruct-Preview-v0.1")
class RWKV6Qwen2Model(Rwkv6Model):
model_arch = gguf.MODEL_ARCH.RWKV6QWEN2
@@ -136,6 +138,7 @@ class RWKV6Qwen2Model(Rwkv6Model):
@ModelBase.register("Rwkv7ForCausalLM", "RWKV7ForCausalLM")
@ModelBase.example("fla-hub/rwkv7-1.5B-world")
class Rwkv7Model(TextModel):
model_arch = gguf.MODEL_ARCH.RWKV7
@@ -261,6 +264,7 @@ class Rwkv7Model(TextModel):
@ModelBase.register("RwkvHybridForCausalLM")
@ModelBase.example("RWKV-Red-Team/ARWKV-7B-Preview-0.1")
class ARwkv7Model(Rwkv7Model):
model_arch = gguf.MODEL_ARCH.ARWKV7
+2
View File
@@ -12,6 +12,7 @@ from .qwenvl import Qwen2VLVisionModel
@ModelBase.register("Sarashina2VisionForCausalLM")
@ModelBase.example("sbintuitions/sarashina2.2-vision-3b")
class Sarashina2VLTextModel(LlamaModel):
model_arch = gguf.MODEL_ARCH.LLAMA
@@ -26,6 +27,7 @@ class Sarashina2VLTextModel(LlamaModel):
@ModelBase.register("Sarashina2VisionForCausalLM")
@ModelBase.example("sbintuitions/sarashina2.2-vision-3b")
class Sarashina2VLVisionModel(Qwen2VLVisionModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
+1
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("SmallThinkerForCausalLM")
@ModelBase.example("PowerInfer/SmallThinker-4BA0.6B-Instruct")
class SmallThinkerModel(TextModel):
model_arch = gguf.MODEL_ARCH.SMALLTHINKER
+1
View File
@@ -9,6 +9,7 @@ from .base import MmprojModel, ModelBase, gguf
@ModelBase.register("Idefics3ForConditionalGeneration", "SmolVLMForConditionalGeneration")
@ModelBase.example("HuggingFaceTB/SmolVLM-Instruct", "HuggingFaceM4/Idefics3-8B-Llama3")
class SmolVLMModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
+1
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("StableLmForCausalLM", "StableLMEpochForCausalLM", "LlavaStableLMEpochForCausalLM")
@ModelBase.example("stabilityai/stablelm-2-1_6b")
class StableLMModel(TextModel):
model_arch = gguf.MODEL_ARCH.STABLELM
+2
View File
@@ -4,6 +4,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("GPTBigCodeForCausalLM")
@ModelBase.example("bigcode/gpt_bigcode-santacoder")
class StarCoderModel(TextModel):
model_arch = gguf.MODEL_ARCH.STARCODER
@@ -19,5 +20,6 @@ class StarCoderModel(TextModel):
@ModelBase.register("Starcoder2ForCausalLM")
@ModelBase.example("bigcode/starcoder2-3b")
class StarCoder2Model(TextModel):
model_arch = gguf.MODEL_ARCH.STARCODER2
+3
View File
@@ -16,6 +16,7 @@ from .qwen import Qwen3Model
@ModelBase.register("StepVLForConditionalGeneration", "Step3p7ForConditionalGeneration")
@ModelBase.example("stepfun-ai/Step3-VL-10B", "stepfun-ai/Step-3.7-Flash")
class Step3VLVisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
@@ -91,11 +92,13 @@ class Step3VLVisionModel(MmprojModel):
@ModelBase.register("StepVLForConditionalGeneration")
@ModelBase.example("stepfun-ai/Step3-VL-10B")
class Step3VLTextModel(Qwen3Model):
model_arch = gguf.MODEL_ARCH.QWEN3
@ModelBase.register("Step3p5ForCausalLM", "Step3p7ForConditionalGeneration")
@ModelBase.example("stepfun-ai/Step-3.7-Flash")
class Step35Model(TextModel):
model_arch = gguf.MODEL_ARCH.STEP35
supports_mtp_export = True
+2
View File
@@ -16,6 +16,7 @@ from .base import ModelBase, SentencePieceTokenTypes, TextModel, gguf, logger
@ModelBase.register("MT5ForConditionalGeneration")
@ModelBase.register("UMT5ForConditionalGeneration")
@ModelBase.register("UMT5Model")
@ModelBase.example("google-t5/t5-small", "google/flan-t5-small", "google/umt5-small")
class T5Model(TextModel):
model_arch = gguf.MODEL_ARCH.T5
@@ -153,6 +154,7 @@ class T5Model(TextModel):
@ModelBase.register("T5EncoderModel")
@ModelBase.example("sentence-transformers/sentence-t5-base")
class T5EncoderModel(TextModel):
model_arch = gguf.MODEL_ARCH.T5ENCODER
+1
View File
@@ -11,6 +11,7 @@ from .base import LazyTorchTensor, ModelBase, TextModel, gguf
@ModelBase.register("TalkieForCausalLM")
@ModelBase.example("lewtun/talkie-1930-13b-it-hf")
class TalkieModel(TextModel):
model_arch = gguf.MODEL_ARCH.TALKIE
+7
View File
@@ -9,6 +9,7 @@ from .base import MmprojModel, ModelBase, TextModel, gguf
@ModelBase.register("UltravoxModel")
@ModelBase.example("fixie-ai/ultravox-v0_5-llama-3_2-1b")
class UltravoxModel(TextModel):
model_arch = gguf.MODEL_ARCH.LLAMA # dummy
@@ -18,6 +19,7 @@ class UltravoxModel(TextModel):
@ModelBase.register("GlmasrModel")
@ModelBase.example("zai-org/GLM-ASR-Nano-2512")
class GlmASRWhisperEncoderModel(MmprojModel):
has_vision_encoder = False
has_audio_encoder = True
@@ -82,6 +84,7 @@ class GlmASRWhisperEncoderModel(MmprojModel):
@ModelBase.register("Qwen2AudioForConditionalGeneration")
@ModelBase.example("Qwen/Qwen2-Audio-7B-Instruct")
class WhisperEncoderModel(MmprojModel):
has_vision_encoder = False # no vision encoder
has_audio_encoder = True
@@ -123,6 +126,7 @@ class WhisperEncoderModel(MmprojModel):
@ModelBase.register("UltravoxModel")
@ModelBase.example("fixie-ai/ultravox-v0_5-llama-3_2-1b")
class UltravoxWhisperEncoderModel(WhisperEncoderModel):
has_vision_encoder = False # no vision encoder
has_audio_encoder = True
@@ -134,6 +138,7 @@ class UltravoxWhisperEncoderModel(WhisperEncoderModel):
@ModelBase.register("MERaLiON2ForConditionalGeneration")
@ModelBase.example("MERaLiON/MERaLiON-2-3B")
class MERaLiONWhisperEncoderModel(WhisperEncoderModel):
has_vision_encoder = False
has_audio_encoder = True
@@ -180,6 +185,7 @@ class MERaLiONWhisperEncoderModel(WhisperEncoderModel):
@ModelBase.register("VoxtralForConditionalGeneration")
@ModelBase.example("mistralai/Voxtral-Mini-3B-2507")
class VoxtralWhisperEncoderModel(WhisperEncoderModel):
has_vision_encoder = False # no vision encoder
has_audio_encoder = True
@@ -191,6 +197,7 @@ class VoxtralWhisperEncoderModel(WhisperEncoderModel):
@ModelBase.register("AudioFlamingo3ForConditionalGeneration")
@ModelBase.example("nvidia/audio-flamingo-3-hf")
class AudioFlamingo3WhisperEncoderModel(WhisperEncoderModel):
def set_gguf_parameters(self):
super().set_gguf_parameters()
+1
View File
@@ -9,6 +9,7 @@ from .base import ModelBase, TextModel, gguf, logger
@ModelBase.register("WavTokenizerDec")
@ModelBase.example("novateur/WavTokenizer-large-speech-75token")
class WavTokenizerDecModel(TextModel):
model_arch = gguf.MODEL_ARCH.WAVTOKENIZER_DEC
+1
View File
@@ -11,6 +11,7 @@ from .base import ModelBase, TextModel, gguf
@ModelBase.register("XverseForCausalLM")
@ModelBase.example("xverse/XVERSE-7B")
class XverseModel(TextModel):
model_arch = gguf.MODEL_ARCH.XVERSE
+1
View File
@@ -9,6 +9,7 @@ from .base import MmprojModel, ModelBase, gguf, logger
@ModelBase.register("YoutuVLForConditionalGeneration")
@ModelBase.example("tencent/Youtu-VL-4B-Instruct")
class YoutuVLVisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
+4
View File
@@ -29,6 +29,7 @@ The required steps to implement for an HF model are:
```python
@ModelBase.register("MyModelForCausalLM")
@ModelBase.example("user/model")
class MyModel(TextModel):
model_arch = gguf.MODEL_ARCH.MYMODEL
```
@@ -37,10 +38,13 @@ or
```python
@ModelBase.register("MyModelForConditionalGeneration")
@ModelBase.example("user/model")
class MyModel(MmprojModel):
model_arch = gguf.MODEL_ARCH.MYMODEL
```
The `example` should point to a valid Hugging Face model that will be used for testing. You can add multiple models if necessary. Prefer a non-gated model, or tiny random weights if no such model exists.
2. Define the layout of the GGUF tensors in [constants.py](/gguf-py/gguf/constants.py)
Add an enum entry in `MODEL_ARCH`, the model human friendly name in `MODEL_ARCH_NAMES` and the GGUF tensor names in `MODEL_TENSORS`.