from datasets import load_dataset
from transformers import AutoProcessor, Llama4ForConditionalGeneration
from llmcompressor.modifiers.quantization import GPTQModifier
from llmcompressor import oneshot
import argparse
from compressed_tensors.quantization import QuantizationScheme, QuantizationArgs, QuantizationType, QuantizationStrategy
defparse_actorder(value):
# Interpret the input value for --actorderif value.lower() == "false":
returnFalseelif value.lower() == "group":
return"group"elif value.lower() == "weight":
return"weight"else:
raise argparse.ArgumentTypeError("Invalid value for --actorder. Use 'group', 'weight', or 'False'.")
defparse_sym(value):
if value.lower() == "false":
returnFalseelif value.lower() == "true":
returnTrueelse:
raise argparse.ArgumentTypeError(f"Invalid value for --sym. Use false or true, but got {value}")
parser = argparse.ArgumentParser()
parser.add_argument('--model_path', type=str, required=True)
parser.add_argument('--quant_path', type=str, required=True)
parser.add_argument('--group_size', type=int, required=True)
parser.add_argument('--calib_size', type=int, required=True)
parser.add_argument('--dampening_frac', type=float, required=True)
parser.add_argument('--observer', type=str, required=True) # mse or minmax
parser.add_argument('--sym', type=parse_sym, required=True) # true or false
parser.add_argument('--actorder', type=parse_actorder, required=True) # group or weight or false
parser.add_argument('--pipeline', type=str, default="basic") # ['basic', 'datafree', 'sequential', independent]
args = parser.parse_args()
model = Llama4ForConditionalGeneration.from_pretrained(
args.model_path,
torch_dtype="auto",
trust_remote_code=True,
)
processor = AutoProcessor.from_pretrained(args.model_path, trust_remote_code=True)
defpreprocess_fn(example):
# prepare for multimodal processorfor msg in example["messages"]:
msg["content"] = [{'type': 'text', 'text': msg['content']}]
return {"text": processor.apply_chat_template(example["messages"], add_generation_prompt=False, tokenize=False)}
ds = load_dataset("neuralmagic/LLM_compression_calibration", split="train")
ds = ds.map(preprocess_fn)
print(f"================================================================================")
print(f"[For debugging] Calibration data sample is:\n{repr(ds[0]['text'])}")
print(f"================================================================================")
quant_scheme = QuantizationScheme(
targets=["Linear"],
weights=QuantizationArgs(
num_bits=4,
type=QuantizationType.INT,
symmetric=args.sym,
group_size=args.group_size,
strategy=QuantizationStrategy.GROUP,
observer=args.observer,
actorder=args.actorder
),
input_activations=None,
output_activations=None,
)
recipe = [
GPTQModifier(
targets=["Linear"],
ignore=[
"re:.*lm_head",
"re:.*multi_modal_projector",
"re:.*vision_model",
],
dampening_frac=args.dampening_frac,
config_groups={"group_0": quant_scheme},
)
]
oneshot(
model=model,
dataset=ds,
recipe=recipe,
num_calibration_samples=args.calib_size,
max_seq_length=4096,
pipeline=args.pipeline,
)
SAVE_DIR = args.quant_path
model.save_pretrained(SAVE_DIR)
print(f"Model saved to {SAVE_DIR}. Please manually copy other files like tokenizer, proprocessors, etc.")
Runs of RedHatAI Llama-Guard-4-12B-quantized.w4a16 on huggingface.co
2.2K
Total runs
0
24-hour runs
218
3-day runs
203
7-day runs
-1.2K
30-day runs
More Information About Llama-Guard-4-12B-quantized.w4a16 huggingface.co Model
Llama-Guard-4-12B-quantized.w4a16 huggingface.co
Llama-Guard-4-12B-quantized.w4a16 huggingface.co is an AI model on huggingface.co that provides Llama-Guard-4-12B-quantized.w4a16's model effect (), which can be used instantly with this RedHatAI Llama-Guard-4-12B-quantized.w4a16 model. huggingface.co supports a free trial of the Llama-Guard-4-12B-quantized.w4a16 model, and also provides paid use of the Llama-Guard-4-12B-quantized.w4a16. Support call Llama-Guard-4-12B-quantized.w4a16 model through api, including Node.js, Python, http.
Llama-Guard-4-12B-quantized.w4a16 huggingface.co is an online trial and call api platform, which integrates Llama-Guard-4-12B-quantized.w4a16's modeling effects, including api services, and provides a free online trial of Llama-Guard-4-12B-quantized.w4a16, you can try Llama-Guard-4-12B-quantized.w4a16 online for free by clicking the link below.
RedHatAI Llama-Guard-4-12B-quantized.w4a16 online free url in huggingface.co:
Llama-Guard-4-12B-quantized.w4a16 is an open source model from GitHub that offers a free installation service, and any user can find Llama-Guard-4-12B-quantized.w4a16 on GitHub to install. At the same time, huggingface.co provides the effect of Llama-Guard-4-12B-quantized.w4a16 install, users can directly use Llama-Guard-4-12B-quantized.w4a16 installed effect in huggingface.co for debugging and trial. It also supports api for free installation.
Llama-Guard-4-12B-quantized.w4a16 install url in huggingface.co: