diff --git a/README.md b/README.md
index b1b3d01..aa33951 100644
--- a/README.md
+++ b/README.md
@@ -1,12 +1,13 @@
# MSQuant
-Model Quantization Tool with NiceGUI interface for AWQ and NVFP4 quantization methods.
+Model Quantization Tool with NiceGUI interface for AWQ, NVFP4, and GGUF quantization methods.
## Features
- **Quantization Methods**
- AWQ (Activation-aware Weight Quantization): 4-bit integer quantization
- NVFP4 (NVIDIA FP4): 4-bit floating-point quantization
+ - GGUF (GGML Universal File): Multiple quantization levels (Q4_K_M, Q5_K_M, Q6_K, Q8_0, etc.) using llama.cpp
- **Web Interface**
- Real-time GPU monitoring with visual charts (Highcharts)
@@ -118,11 +119,21 @@ pixi install
Navigate to the **Configure** page and set:
- Model ID (e.g., `meta-llama/Llama-3.1-8B`)
-- Quantization method (AWQ or NVFP4)
-- Calibration dataset settings
-- Method-specific parameters
-
-**Note:** Output format is determined by the quantization backend. The saved checkpoint format follows the backend's default conventions.
+- Quantization method (AWQ, NVFP4, or GGUF)
+- Calibration dataset settings (required for AWQ and NVFP4)
+- Method-specific parameters:
+ - **AWQ**: Weight bits, group size, zero point
+ - **NVFP4**: Activation/weight schemes
+ - **GGUF**: Quantization type (Q4_K_M recommended, Q5_K_M for best quality), intermediate format (f16 default)
+
+**Note:**
+- AWQ and NVFP4 output formats follow llmcompressor conventions (binary or safetensors)
+- GGUF produces `.gguf` files compatible with llama.cpp, Ollama, and other GGUF-compatible inference engines
+- GGUF quantization types:
+ - **Q4_K_M**: Recommended for balanced quality and size
+ - **Q5_K_M**: Best quality while maintaining reasonable size
+ - **Q6_K, Q8_0**: Higher precision options
+ - **Q2_K, Q3_K**: Smaller sizes with reduced quality
### 2. Monitor Progress
@@ -176,6 +187,7 @@ Replace `OWNER` with your GitHub username/organization.
Built with:
- [NiceGUI](https://nicegui.io/) - Web interface
-- [llmcompressor](https://github.com/vllm-project/llm-compressor) - Quantization engine
+- [llmcompressor](https://github.com/vllm-project/llm-compressor) - Quantization engine for AWQ/NVFP4
+- [llama.cpp](https://github.com/ggerganov/llama.cpp) - GGUF quantization and inference
- [vLLM](https://github.com/vllm-project/vllm) - LLM inference
- [Pixi](https://pixi.sh/) - Package management
diff --git a/docker/Dockerfile.gpu b/docker/Dockerfile.gpu
index 7fd7dc9..474a6f5 100644
--- a/docker/Dockerfile.gpu
+++ b/docker/Dockerfile.gpu
@@ -32,6 +32,25 @@ RUN pip install "transformers>=4.52.0"
RUN pip install "llmcompressor"
+# Install llama.cpp for GGUF quantization support
+ARG LLAMA_CPP_VERSION=b6945
+RUN apt-get update && apt-get install -y --no-install-recommends \
+ wget \
+ unzip \
+ && apt-get clean && \
+ rm -rf /var/lib/apt/lists/* /var/cache/apt/archives/*
+
+# Download and install pre-compiled llama.cpp binary
+RUN wget -q https://github.com/ggml-org/llama.cpp/releases/download/${LLAMA_CPP_VERSION}/llama-${LLAMA_CPP_VERSION}-bin-ubuntu-x64.zip -O /tmp/llama.zip && \
+ unzip -q /tmp/llama.zip -d /opt && \
+ mv /opt/llama-${LLAMA_CPP_VERSION}-bin-ubuntu-x64 /opt/llama.cpp && \
+ rm /tmp/llama.zip && \
+ chmod +x /opt/llama.cpp/llama-* && \
+ pip install gguf
+
+# Add llama.cpp to PATH
+ENV PATH="/opt/llama.cpp:${PATH}"
+
WORKDIR /workspace
# Copy source
diff --git a/pixi.toml b/pixi.toml
index efd3edb..5e1e34b 100644
--- a/pixi.toml
+++ b/pixi.toml
@@ -40,6 +40,8 @@ huggingface_hub = "*"
hf_transfer = "*"
tqdm = "*"
pandas = ">=2.2.0"
+# GGUF support
+gguf = "*"
# Optional dev extras
types-tqdm = "*"
diff --git a/src/msquant/app/pages/configure.py b/src/msquant/app/pages/configure.py
index 113e39b..5bed9fb 100644
--- a/src/msquant/app/pages/configure.py
+++ b/src/msquant/app/pages/configure.py
@@ -20,6 +20,8 @@ def create_configure_page(job_service: JobService, storage_service: StorageServi
'w_bit': 4,
'group_size': 128,
'zero_point': True,
+ 'gguf_quant_type': 'Q4_K_M',
+ 'gguf_intermediate_format': 'f16',
}
def start_quantization():
@@ -37,6 +39,8 @@ def start_quantization():
w_bit=form_data['w_bit'],
group_size=form_data['group_size'],
zero_point=form_data['zero_point'],
+ gguf_quant_type=form_data['gguf_quant_type'],
+ gguf_intermediate_format=form_data['gguf_intermediate_format'],
)
if job_service.start_job(config):
@@ -61,7 +65,7 @@ def start_quantization():
# Quantization method
ui.label('Quantization Method').classes('text-2xl font-bold')
with ui.row().classes('w-full gap-4'):
- ui.radio(['awq', 'nvfp4'], value='awq').bind_value(form_data, 'quant_method').props('inline')
+ ui.radio(['awq', 'nvfp4', 'gguf'], value='awq').bind_value(form_data, 'quant_method').props('inline')
ui.separator()
@@ -99,9 +103,32 @@ def start_quantization():
label='Group Size'
).classes('flex-1').bind_value(form_data, 'group_size').props('use-input new-value-mode=add-unique')
ui.checkbox('Zero Point', value=True).bind_value(form_data, 'zero_point')
-
+
ui.separator()
-
+
+ # GGUF-specific settings
+ ui.label('GGUF Settings (only for GGUF method)').classes('text-2xl font-bold')
+ with ui.row().classes('w-full gap-4'):
+ ui.select(
+ ['Q2_K', 'Q3_K_S', 'Q3_K_M', 'Q3_K_L', 'Q4_0', 'Q4_1', 'Q4_K_S', 'Q4_K_M',
+ 'Q5_0', 'Q5_1', 'Q5_K_S', 'Q5_K_M', 'Q6_K', 'Q8_0', 'F16', 'F32'],
+ value='Q4_K_M',
+ label='Quantization Type'
+ ).classes('flex-1').bind_value(form_data, 'gguf_quant_type')
+ ui.select(
+ ['f16', 'f32', 'q8_0'],
+ value='f16',
+ label='Intermediate Format'
+ ).classes('flex-1').bind_value(form_data, 'gguf_intermediate_format')
+ ui.html('''
+
+ Recommended: Q4_K_M (balanced), Q5_K_M (best quality)
+ Intermediate: f16 (default), f32 (higher precision), q8_0 (smaller)
+
+ ''', sanitize=False)
+
+ ui.separator()
+
# Actions
with ui.row().classes('w-full gap-4 justify-end'):
ui.button('Start Quantization', on_click=start_quantization).props('color=primary size=lg')
diff --git a/src/msquant/core/quantizer/config.py b/src/msquant/core/quantizer/config.py
index 0a18c1a..145a04e 100644
--- a/src/msquant/core/quantizer/config.py
+++ b/src/msquant/core/quantizer/config.py
@@ -24,6 +24,9 @@ def __init__(
w_scheme: str = "fp4",
non_uniform: bool = False,
mix_fp8: bool = False,
+ # GGUF params
+ gguf_quant_type: str = "Q4_K_M",
+ gguf_intermediate_format: str = "f16",
# Paths
hf_home: str = "/workspace/hf",
hf_datasets_cache: str = "/workspace/hf/datasets",
@@ -71,7 +74,16 @@ def __init__(
self.w_scheme = w_scheme
self.non_uniform = non_uniform
self.mix_fp8 = mix_fp8
-
+
+ # GGUF - validate and convert types
+ if not isinstance(gguf_quant_type, str):
+ raise TypeError(f"gguf_quant_type must be a string, got {type(gguf_quant_type).__name__}")
+ self.gguf_quant_type = gguf_quant_type.upper()
+
+ if not isinstance(gguf_intermediate_format, str):
+ raise TypeError(f"gguf_intermediate_format must be a string, got {type(gguf_intermediate_format).__name__}")
+ self.gguf_intermediate_format = gguf_intermediate_format.lower()
+
# Paths
self.hf_home = hf_home
self.hf_datasets_cache = hf_datasets_cache
@@ -89,15 +101,36 @@ def validate(self):
"""Validate configuration."""
if not self.model_id:
raise ValueError("model_id is required")
-
- if self.quant_method not in ["awq", "nvfp4"]:
+
+ if self.quant_method not in ["awq", "nvfp4", "gguf"]:
raise ValueError(f"Invalid quant_method: {self.quant_method}")
-
+
if self.output_format not in ["binary", "safetensors"]:
raise ValueError(f"Invalid output_format: {self.output_format}")
-
+
if self.quant_method == "awq":
if self.w_bit not in [2, 3, 4, 5, 8]:
raise ValueError(f"Invalid w_bit for AWQ: {self.w_bit}")
if self.group_size <= 0:
raise ValueError(f"Invalid group_size: {self.group_size}")
+
+ if self.quant_method == "gguf":
+ # Validate GGUF quantization type
+ valid_gguf_types = [
+ "Q2_K", "Q3_K_S", "Q3_K_M", "Q3_K_L",
+ "Q4_0", "Q4_1", "Q4_K_S", "Q4_K_M",
+ "Q5_0", "Q5_1", "Q5_K_S", "Q5_K_M",
+ "Q6_K", "Q8_0", "F16", "F32"
+ ]
+ if self.gguf_quant_type not in valid_gguf_types:
+ raise ValueError(
+ f"Invalid gguf_quant_type: {self.gguf_quant_type}. "
+ f"Must be one of: {', '.join(valid_gguf_types)}"
+ )
+
+ # Validate intermediate format
+ if self.gguf_intermediate_format not in ["f16", "f32", "q8_0"]:
+ raise ValueError(
+ f"Invalid gguf_intermediate_format: {self.gguf_intermediate_format}. "
+ f"Must be one of: f16, f32, q8_0"
+ )
diff --git a/src/msquant/core/quantizer/engine.py b/src/msquant/core/quantizer/engine.py
index a21d486..79b97dc 100644
--- a/src/msquant/core/quantizer/engine.py
+++ b/src/msquant/core/quantizer/engine.py
@@ -240,6 +240,222 @@ def run(config: QuantizationConfig, logger: QuantizationLogger):
return config.out_dir
+class GGUFQuantizer:
+ """GGUF quantization handler using llama.cpp."""
+
+ @staticmethod
+ def _check_llama_cpp_available():
+ """Check if llama.cpp tools are available."""
+ try:
+ # Check if convert-hf-to-gguf.py is available
+ result = subprocess.run(
+ ["bash", "-lc", "which python"],
+ capture_output=True,
+ text=True,
+ check=False
+ )
+ if result.returncode != 0:
+ raise RuntimeError("Python not found in PATH")
+
+ # Check if llama-quantize is available
+ result = subprocess.run(
+ ["bash", "-lc", "which llama-quantize"],
+ capture_output=True,
+ text=True,
+ check=False
+ )
+ if result.returncode != 0:
+ raise RuntimeError(
+ "llama-quantize not found in PATH. "
+ "Please ensure llama.cpp is properly installed."
+ )
+ return True
+ except Exception as e:
+ raise RuntimeError(f"llama.cpp availability check failed: {e}") from e
+
+ @staticmethod
+ def _download_model(model_id: str, cache_dir: str, logger: QuantizationLogger) -> str:
+ """Download HuggingFace model to local cache."""
+ from huggingface_hub import snapshot_download
+
+ logger.info(f"Downloading model {model_id} to cache...")
+ try:
+ local_path = snapshot_download(
+ repo_id=model_id,
+ cache_dir=cache_dir,
+ local_dir_use_symlinks=False
+ )
+ logger.info(f"Model downloaded to {local_path}")
+ return local_path
+ except Exception as e:
+ raise RuntimeError(f"Failed to download model: {e}") from e
+
+ @staticmethod
+ def _convert_to_gguf_intermediate(
+ model_path: str,
+ output_file: str,
+ intermediate_format: str,
+ logger: QuantizationLogger
+ ):
+ """Convert HuggingFace model to intermediate GGUF format."""
+ logger.info(f"Converting model to GGUF {intermediate_format} format...")
+
+ # Find the convert-hf-to-gguf.py script
+ convert_script = "/opt/llama.cpp/convert-hf-to-gguf.py"
+
+ # Build the conversion command
+ cmd = [
+ "python",
+ convert_script,
+ model_path,
+ "--outfile", output_file,
+ "--outtype", intermediate_format
+ ]
+
+ logger.info(f"Running command: {' '.join(cmd)}")
+
+ try:
+ process = subprocess.Popen(
+ cmd,
+ stdout=subprocess.PIPE,
+ stderr=subprocess.STDOUT,
+ text=True,
+ bufsize=1
+ )
+
+ # Stream output
+ if process.stdout:
+ for line in process.stdout:
+ line = line.rstrip()
+ if line:
+ logger.info(f"[convert] {line}")
+
+ process.wait()
+
+ if process.returncode != 0:
+ raise RuntimeError(f"Conversion failed with exit code {process.returncode}")
+
+ logger.info(f"Conversion completed. Intermediate GGUF file: {output_file}")
+
+ except Exception as e:
+ raise RuntimeError(f"GGUF conversion failed: {e}") from e
+
+ @staticmethod
+ def _quantize_gguf(
+ input_file: str,
+ output_file: str,
+ quant_type: str,
+ logger: QuantizationLogger
+ ):
+ """Quantize GGUF file to target precision."""
+ logger.info(f"Quantizing GGUF to {quant_type}...")
+
+ # Skip quantization if target format is already F16 or F32
+ if quant_type in ["F16", "F32"]:
+ logger.info(f"Target format {quant_type} matches intermediate format, skipping quantization")
+ # Copy the file instead
+ import shutil
+ shutil.copy2(input_file, output_file)
+ return
+
+ # Build the quantization command
+ cmd = [
+ "llama-quantize",
+ input_file,
+ output_file,
+ quant_type
+ ]
+
+ logger.info(f"Running command: {' '.join(cmd)}")
+
+ try:
+ process = subprocess.Popen(
+ cmd,
+ stdout=subprocess.PIPE,
+ stderr=subprocess.STDOUT,
+ text=True,
+ bufsize=1
+ )
+
+ # Stream output
+ if process.stdout:
+ for line in process.stdout:
+ line = line.rstrip()
+ if line:
+ logger.info(f"[quantize] {line}")
+
+ process.wait()
+
+ if process.returncode != 0:
+ raise RuntimeError(f"Quantization failed with exit code {process.returncode}")
+
+ logger.info(f"Quantization completed. Output file: {output_file}")
+
+ except Exception as e:
+ raise RuntimeError(f"GGUF quantization failed: {e}") from e
+
+ @staticmethod
+ def run(config: QuantizationConfig, logger: QuantizationLogger):
+ """Run GGUF quantization."""
+ import os
+
+ # Check if llama.cpp is available
+ GGUFQuantizer._check_llama_cpp_available()
+
+ logger.info("GGUF Quantization config:", {
+ "quant_type": config.gguf_quant_type,
+ "intermediate_format": config.gguf_intermediate_format,
+ "model_id": config.model_id,
+ })
+
+ t0 = time.time()
+
+ # Step 1: Download model from HuggingFace
+ model_path = GGUFQuantizer._download_model(
+ config.model_id,
+ config.hf_home,
+ logger
+ )
+
+ # Step 2: Convert to intermediate GGUF format
+ os.makedirs(config.out_dir, exist_ok=True)
+ safe_name = config.model_id.split("/")[-1].replace(":", "-")
+ intermediate_file = os.path.join(
+ config.out_dir,
+ f"{safe_name}-{config.gguf_intermediate_format}.gguf"
+ )
+
+ GGUFQuantizer._convert_to_gguf_intermediate(
+ model_path,
+ intermediate_file,
+ config.gguf_intermediate_format,
+ logger
+ )
+
+ # Step 3: Quantize to target precision
+ final_file = os.path.join(
+ config.out_dir,
+ f"{safe_name}-{config.gguf_quant_type}.gguf"
+ )
+
+ GGUFQuantizer._quantize_gguf(
+ intermediate_file,
+ final_file,
+ config.gguf_quant_type,
+ logger
+ )
+
+ # Clean up intermediate file if different from final
+ if intermediate_file != final_file and os.path.exists(intermediate_file):
+ logger.info(f"Cleaning up intermediate file: {intermediate_file}")
+ os.remove(intermediate_file)
+
+ dt = time.time() - t0
+ logger.info(f"Completed. Saved GGUF quantized model to {final_file} in {dt:.1f}s")
+
+ return config.out_dir
+
+
def quantize(config: QuantizationConfig, log_callback: Optional[Callable[[str], None]] = None):
"""
Main quantization entry point.
@@ -285,6 +501,8 @@ def quantize(config: QuantizationConfig, log_callback: Optional[Callable[[str],
output_dir = AWQQuantizer.run(config, logger)
elif config.quant_method == "nvfp4":
output_dir = NVFP4Quantizer.run(config, logger)
+ elif config.quant_method == "gguf":
+ output_dir = GGUFQuantizer.run(config, logger)
else:
raise ValueError(f"Unsupported quantization method: {config.quant_method}")