Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions conversion/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@
"ChameleonForConditionalGeneration": "chameleon",
"ChatGLMForConditionalGeneration": "chatglm",
"ChatGLMModel": "chatglm",
"ClefModel": "clef",
"CodeShellForCausalLM": "codeshell",
"CogVLMForCausalLM": "cogvlm",
"Cohere2MoeForCausalLM": "command_r",
Expand Down Expand Up @@ -271,6 +272,7 @@

MMPROJ_MODEL_MAP: dict[str, str] = {
"AudioFlamingo3ForConditionalGeneration": "ultravox",
"ClefModel": "clef",
"CogVLMForCausalLM": "cogvlm",
"DeepseekOCR2ForCausalLM": "deepseek",
"DeepseekOCRForCausalLM": "deepseek",
Expand Down
150 changes: 150 additions & 0 deletions conversion/clef.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,150 @@
from __future__ import annotations

import json
import math

from pathlib import Path
from typing import Any, Iterable, Iterator, TYPE_CHECKING

import torch

if TYPE_CHECKING:
from torch import Tensor

from .base import MmprojModel, ModelBase, gguf, logger
from .qwen import Qwen3_5TextModel


def _is_clef_checkpoint(dir_model: Path) -> bool:
return (dir_model / "joint_head_config.json").is_file() and (dir_model / "config.json").is_file()


@ModelBase.register_hparams_loader(_is_clef_checkpoint)
def _load_clef_hparams(dir_model: Path) -> dict[str, Any]:
logger.info("gguf: detected Clef checkpoint")
hparams = ModelBase.load_hparams(dir_model, False, guess=False)
hparams["architectures"] = ["ClefModel"]
with open(dir_model / "joint_head_config.json", encoding="utf-8") as f:
hparams["decision"] = json.load(f)
return hparams


@ModelBase.register("ClefModel")
class ClefModel(Qwen3_5TextModel):
model_arch = gguf.MODEL_ARCH.CLEF
no_mtp = True # the checkpoint has no MTP head

# prompt follows joint_schema_model.py of the model repo
_SYSTEM_PROMPT = (
"Read the complete state and schema. Decide every field jointly. Each answer "
"must be exactly one of that field's allowed options."
)
# torch.nn.LayerNorm default, used by the head
_HEAD_NORM_EPS = 1e-5

def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
head = self.hparams["decision"]
self._n_routing = head["routing_layers"]
# the head blocks are named dec.blk.N, routing blocks first
self.tensor_map = gguf.get_tensor_name_map(self.model_arch, max(self.block_count, self._n_routing + head["layers"]))
self._scales: dict[str, float] = {}

def set_vocab(self):
super().set_vocab()
self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])

@classmethod
def _systemone_template(cls) -> str:
def text(value: str) -> str:
return "{{ " + json.dumps(value) + " }}"

def render(name: str) -> str:
# strings are used as is, other values are compact JSON
return "{{ " + name + " if " + name + " is string else " + name + " | tojson(separators=[',', ':']) }}"

# the pieces of the prompt are tokenized one by one, the server gives the text that separates them (sep)
# and the text that starts the span of a question or of an option (mark_question, mark_option)
# the keys of JSON objects are given in sorted order
option = (
"{% set d = o.description %}"
"{% if q.type == 'noul' and d is none %}"
"{% set d = 'The proposition is true or the answer is yes.' if o.key == 'true' else 'The proposition is false or the answer is no.' %}"
"{% endif %}"
"{{ ({'option_id': o.key} if d is none else {'description': d, 'option_id': o.key}) | tojson(separators=[',', ':']) }}"
)
return (
text(f"<|im_start|>system\n{cls._SYSTEM_PROMPT}<|im_end|>\n<|im_start|>user\nSTATE:\n")
+ "{{ sep }}" + render("state")
+ "{{ sep }}" + text("\n\nSCHEMA FIELDS:\n")
+ "{% for q in questions %}"
+ "{{ sep }}" + text("\nFIELD ") + "{{ loop.index }}" + text("\nID: ") + "{{ q.id }}"
+ text("\nTYPE: ") + "{{ q.type }}" + text("\nINSTRUCTION: ")
+ "{{ sep }}{{ mark_question }}" + render("q.instructions")
+ "{{ sep }}" + text("\nALLOWED OPTIONS:\n")
+ "{% for o in q.options %}"
+ "{{ sep }}" + text("OPTION ") + "{{ loop.index }}" + text(": ")
+ "{{ sep }}{{ mark_option }}" + option
+ "{{ sep }}" + text("\n")
+ "{% endfor %}"
+ "{{ sep }}" + text("END FIELD\n")
+ "{% endfor %}"
+ "{{ sep }}" + text("\n<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\nJOINT SCHEMA DECISIONS:")
)

def set_gguf_parameters(self):
super().set_gguf_parameters()
head = self.hparams["decision"]
self.gguf_writer.add_decision_type(gguf.DecisionType.CLEF)
self.gguf_writer.add_decision_routing_block_count(head["routing_layers"])
self.gguf_writer.add_decision_block_count(head["layers"])
self.gguf_writer.add_decision_head_count(head["heads"])
self.gguf_writer.add_layer_norm_eps(self._HEAD_NORM_EPS)

def get_tensors(self) -> Iterator[tuple[str, Tensor]]:
yield from super().get_tensors()
from safetensors.torch import load_file
for name, data in load_file(self.dir_model / "joint_head.safetensors").items():
yield "joint_head." + name, data

def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
if not name.startswith("joint_head."):
yield from super().modify_tensors(data_torch, name, bid)
return

parts = name.split(".")

# learned scalars, stored as the values used at inference
if len(parts) == 2 and data_torch.ndim == 0:
value = float(data_torch)
if parts[1] == "residual_gate":
self._scales[parts[1]] = 1.0 / (1.0 + math.exp(-value))
else:
self._scales[parts[1]] = math.exp(min(value, math.log(100.0)))
if len(self._scales) == 3:
scales = [self._scales[k] for k in ("prior_logit_scale", "joint_logit_scale", "residual_gate")]
yield self.format_tensor_name(gguf.MODEL_TENSOR.DECISION_SCALES, suffix=""), torch.tensor(scales, dtype=torch.float32)
return

# routing blocks come first
if parts[1] == "layers":
parts[2] = str(int(parts[2]) + self._n_routing)
name = ".".join(parts)

# nn.MultiheadAttention keeps q, k, v in one tensor
for suffix in ("weight", "bias"):
if name.endswith(".in_proj_" + suffix):
prefix = name[:-len("in_proj_" + suffix)]
for x, data in zip("qkv", data_torch.chunk(3, dim=0)):
yield self.map_tensor_name(prefix + x + "." + suffix), data
return

yield self.map_tensor_name(name), data_torch


@ModelBase.register("ClefModel")
class ClefVisionModel(MmprojModel):
def __init__(self, *args, **kwargs):
del args, kwargs
raise NotImplementedError(
"multimodal input is not supported yet for Clef, requires https://github.com/ggml-org/llama.cpp/pull/29622 to be merged first")
101 changes: 101 additions & 0 deletions gguf-py/gguf/constants.py
Original file line number Diff line number Diff line change
Expand Up @@ -280,6 +280,15 @@ class Classifier:
class ShortConv:
L_CACHE = "{arch}.shortconv.l_cache"

class Decision:
TYPE = "{arch}.decision.type"
# note: single-use-case keys can be hard-coded in cpp code
BLOCK_COUNT = "{arch}.decision.block_count"
ROUTING_BLOCK_COUNT = "{arch}.decision.routing_block_count"
HEAD_COUNT = "{arch}.decision.head_count"
MAX_HEAD_TOKENS = "{arch}.decision.max_head_tokens"
TEMPERATURE = "{arch}.decision.temperature.{name}" # name: "<type>" or "<type>.<n_opt bucket>"

class Tokenizer:
MODEL = "tokenizer.ggml.model"
PRE = "tokenizer.ggml.pre"
Expand Down Expand Up @@ -486,6 +495,7 @@ class MODEL_ARCH(IntEnum):
QWEN3VLMOE = auto()
QWEN35 = auto()
QWEN35MOE = auto()
CLEF = auto()
PHI2 = auto()
PHI3 = auto()
PHIMOE = auto()
Expand Down Expand Up @@ -783,6 +793,7 @@ class MODEL_TENSOR(IntEnum):
DEC_ATTN_OUT = auto()
DEC_ATTN_REL_B = auto()
DEC_CROSS_ATTN_NORM = auto()
DEC_CROSS_ATTN_NORM_KV = auto()
DEC_CROSS_ATTN_Q = auto()
DEC_CROSS_ATTN_K = auto()
DEC_CROSS_ATTN_V = auto()
Expand All @@ -807,6 +818,19 @@ class MODEL_TENSOR(IntEnum):
CLS = auto() # classifier
CLS_OUT = auto() # classifier output projection
CLS_NORM = auto()
DECISION_HIDDEN_NORM = auto()
DECISION_PROJ_MEMORY = auto()
DECISION_PROJ_QUESTION = auto()
DECISION_PROJ_OPTION_QUESTION = auto()
DECISION_PROJ_GLOBAL = auto()
DECISION_PROJ_OPTION_CONTEXT = auto()
DECISION_PROJ_OPTION_LEXICAL = auto()
DECISION_OPTION_SUMMARY_NORM = auto()
DECISION_FIELD_NORM = auto()
DECISION_OPTION_NORM = auto()
DECISION_SCALES = auto()
DECISION_SCORER = auto()
DECISION_SCORER_OUT = auto()
CONV1D = auto()
CONVNEXT_DW = auto()
CONVNEXT_NORM = auto()
Expand Down Expand Up @@ -1202,6 +1226,7 @@ class MODEL_TENSOR(IntEnum):
MODEL_ARCH.QWEN3VLMOE: "qwen3vlmoe",
MODEL_ARCH.QWEN35: "qwen35",
MODEL_ARCH.QWEN35MOE: "qwen35moe",
MODEL_ARCH.CLEF: "clef",
MODEL_ARCH.PHI2: "phi2",
MODEL_ARCH.PHI3: "phi3",
MODEL_ARCH.PHIMOE: "phimoe",
Expand Down Expand Up @@ -1498,6 +1523,7 @@ class MODEL_TENSOR(IntEnum):
MODEL_TENSOR.DEC_ATTN_OUT: "dec.blk.{bid}.attn_o",
MODEL_TENSOR.DEC_ATTN_REL_B: "dec.blk.{bid}.attn_rel_b",
MODEL_TENSOR.DEC_CROSS_ATTN_NORM: "dec.blk.{bid}.cross_attn_norm",
MODEL_TENSOR.DEC_CROSS_ATTN_NORM_KV: "dec.blk.{bid}.cross_attn_norm_kv",
MODEL_TENSOR.DEC_CROSS_ATTN_Q: "dec.blk.{bid}.cross_attn_q",
MODEL_TENSOR.DEC_CROSS_ATTN_K: "dec.blk.{bid}.cross_attn_k",
MODEL_TENSOR.DEC_CROSS_ATTN_V: "dec.blk.{bid}.cross_attn_v",
Expand All @@ -1522,6 +1548,19 @@ class MODEL_TENSOR(IntEnum):
MODEL_TENSOR.CLS: "cls",
MODEL_TENSOR.CLS_OUT: "cls.output",
MODEL_TENSOR.CLS_NORM: "cls.norm",
MODEL_TENSOR.DECISION_HIDDEN_NORM: "decision.hidden_norm",
MODEL_TENSOR.DECISION_PROJ_MEMORY: "decision.proj_memory",
MODEL_TENSOR.DECISION_PROJ_QUESTION: "decision.proj_question",
MODEL_TENSOR.DECISION_PROJ_OPTION_QUESTION: "decision.proj_option_question",
MODEL_TENSOR.DECISION_PROJ_GLOBAL: "decision.proj_global",
MODEL_TENSOR.DECISION_PROJ_OPTION_CONTEXT: "decision.proj_option_context",
MODEL_TENSOR.DECISION_PROJ_OPTION_LEXICAL: "decision.proj_option_lexical",
MODEL_TENSOR.DECISION_OPTION_SUMMARY_NORM: "decision.option_summary_norm",
MODEL_TENSOR.DECISION_FIELD_NORM: "decision.field_norm",
MODEL_TENSOR.DECISION_OPTION_NORM: "decision.option_norm",
MODEL_TENSOR.DECISION_SCALES: "decision.scales",
MODEL_TENSOR.DECISION_SCORER: "decision.scorer",
MODEL_TENSOR.DECISION_SCORER_OUT: "decision.scorer_out",
MODEL_TENSOR.CONV1D: "conv1d",
MODEL_TENSOR.CONVNEXT_DW: "convnext.{bid}.dw",
MODEL_TENSOR.CONVNEXT_NORM: "convnext.{bid}.norm",
Expand Down Expand Up @@ -2730,6 +2769,60 @@ class MODEL_TENSOR(IntEnum):
MODEL_TENSOR.NEXTN_SHARED_HEAD_HEAD,
MODEL_TENSOR.NEXTN_SHARED_HEAD_NORM,
],
MODEL_ARCH.CLEF: [
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.OUTPUT_NORM,
MODEL_TENSOR.OUTPUT,
MODEL_TENSOR.ATTN_NORM,
MODEL_TENSOR.ATTN_Q,
MODEL_TENSOR.ATTN_Q_NORM,
MODEL_TENSOR.ATTN_K,
MODEL_TENSOR.ATTN_K_NORM,
MODEL_TENSOR.ATTN_V,
MODEL_TENSOR.ATTN_OUT,
MODEL_TENSOR.ATTN_POST_NORM,
MODEL_TENSOR.ATTN_GATE,
MODEL_TENSOR.ATTN_QKV,
MODEL_TENSOR.FFN_GATE,
MODEL_TENSOR.FFN_DOWN,
MODEL_TENSOR.FFN_UP,
MODEL_TENSOR.SSM_A,
MODEL_TENSOR.SSM_CONV1D,
MODEL_TENSOR.SSM_DT,
MODEL_TENSOR.SSM_NORM,
MODEL_TENSOR.SSM_BETA,
MODEL_TENSOR.SSM_ALPHA,
MODEL_TENSOR.SSM_OUT,
# decision head
MODEL_TENSOR.TOKEN_TYPES,
MODEL_TENSOR.DEC_ATTN_NORM,
MODEL_TENSOR.DEC_ATTN_Q,
MODEL_TENSOR.DEC_ATTN_K,
MODEL_TENSOR.DEC_ATTN_V,
MODEL_TENSOR.DEC_ATTN_OUT,
MODEL_TENSOR.DEC_CROSS_ATTN_NORM,
MODEL_TENSOR.DEC_CROSS_ATTN_NORM_KV,
MODEL_TENSOR.DEC_CROSS_ATTN_Q,
MODEL_TENSOR.DEC_CROSS_ATTN_K,
MODEL_TENSOR.DEC_CROSS_ATTN_V,
MODEL_TENSOR.DEC_CROSS_ATTN_OUT,
MODEL_TENSOR.DEC_FFN_NORM,
MODEL_TENSOR.DEC_FFN_DOWN,
MODEL_TENSOR.DEC_FFN_UP,
MODEL_TENSOR.DECISION_HIDDEN_NORM,
MODEL_TENSOR.DECISION_PROJ_MEMORY,
MODEL_TENSOR.DECISION_PROJ_QUESTION,
MODEL_TENSOR.DECISION_PROJ_OPTION_QUESTION,
MODEL_TENSOR.DECISION_PROJ_GLOBAL,
MODEL_TENSOR.DECISION_PROJ_OPTION_CONTEXT,
MODEL_TENSOR.DECISION_PROJ_OPTION_LEXICAL,
MODEL_TENSOR.DECISION_OPTION_SUMMARY_NORM,
MODEL_TENSOR.DECISION_FIELD_NORM,
MODEL_TENSOR.DECISION_OPTION_NORM,
MODEL_TENSOR.DECISION_SCALES,
MODEL_TENSOR.DECISION_SCORER,
MODEL_TENSOR.DECISION_SCORER_OUT,
],
MODEL_ARCH.QWEN35MOE: [
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.OUTPUT_NORM,
Expand Down Expand Up @@ -5362,6 +5455,14 @@ def get_type(val: Any) -> GGUFValueType:
raise ValueError(f"Unknown type: {type(val)}")


class DecisionType:
LAYA = "laya" # head blocks + scorer on the hidden state of one marker token per option
OPENJEV = "openjev" # logits of one label token per option
LEV = "lev" # same as openjev, noul is read from a rating scale
KEV = "kev" # dot product of the hidden states of the last token and of one end token per option
CLEF = "clef" # joint head over all questions, one score per option


class VisionProjectorType:
GEMMA3 = "gemma3"
GEMMA3NV = "gemma3nv"
Expand Down
21 changes: 21 additions & 0 deletions gguf-py/gguf/gguf_writer.py
Original file line number Diff line number Diff line change
Expand Up @@ -1229,6 +1229,27 @@ def add_eom_token_id(self, id: int) -> None:
def add_classifier_output_labels(self, labels: Sequence[str]) -> None:
self.add_array(Keys.Classifier.OUTPUT_LABELS.format(arch=self.arch), labels)

def add_classifier_pooling_type(self, value: PoolingType) -> None:
self.add_uint32(Keys.Classifier.POOLING_TYPE.format(arch=self.arch), value.value)

def add_decision_type(self, value: str) -> None:
self.add_string(Keys.Decision.TYPE.format(arch=self.arch), value)

def add_decision_block_count(self, value: int) -> None:
self.add_uint32(Keys.Decision.BLOCK_COUNT.format(arch=self.arch), value)

def add_decision_routing_block_count(self, value: int) -> None:
self.add_uint32(Keys.Decision.ROUTING_BLOCK_COUNT.format(arch=self.arch), value)

def add_decision_head_count(self, value: int) -> None:
self.add_uint32(Keys.Decision.HEAD_COUNT.format(arch=self.arch), value)

def add_decision_max_head_tokens(self, value: int) -> None:
self.add_uint32(Keys.Decision.MAX_HEAD_TOKENS.format(arch=self.arch), value)

def add_decision_temperature(self, name: str, value: float) -> None:
self.add_float32(Keys.Decision.TEMPERATURE.format(arch=self.arch, name=name), value)

# for vision models

def add_clip_has_vision_encoder(self, value: bool) -> None:
Expand Down
Loading
Loading