Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -104,7 +104,9 @@ The following optional dependencies are available:
- `fp8-infer`: `torchao` package for fp8 inference
- `gptq`: `GPTQModel` package for W4A16 quantization
- `mx`: `microxcaling` package for MX quantization
- `opt`: Shortcut for `fp8`, `gptq`, and `mx` installs
- `data`: `datasets` package for loading calibration data (required by `fms_mo.run_quant` and direct quantization)
- `launcher`: `accelerate` package for the multi-GPU launcher in `build/`
- `opt`: Shortcut for `fp8`, `gptq`, `mx`, `data`, and `launcher` installs
- `aiu`: `ibm-fms` package for AIU model deployment
- `torchvision`: `torch` package for image recognition training and inference
- `triton`: `triton` package for matrix multiplication kernels
Expand Down
10 changes: 9 additions & 1 deletion fms_mo/dq.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,6 @@
import logging

# Third Party
from datasets import load_from_disk
from torch.utils.data import DataLoader
from tqdm import tqdm
from transformers import (
Expand Down Expand Up @@ -51,6 +50,15 @@
from fms_mo.utils.eval_utils import Evaluator, eval_llm_1GPU
from fms_mo.utils.utils import patch_torch_bmm, prepare_input

try:
# Third Party
from datasets import load_from_disk
except ImportError as e:
raise ImportError(
"The `datasets` package is required for data loading. Install it with "
"`pip install fms-model-optimizer[data]`."
) from e

logger = logging.getLogger(__name__)


Expand Down
10 changes: 9 additions & 1 deletion fms_mo/run_quant.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,6 @@
import traceback

# Third Party
from datasets import load_from_disk
from torch.cuda import OutOfMemoryError
from transformers import (
AutoModelForMaskedLM,
Expand Down Expand Up @@ -62,6 +61,15 @@
from fms_mo.utils.import_utils import available_packages
from fms_mo.utils.logging_utils import set_log_level

try:
# Third Party
from datasets import load_from_disk
except ImportError as e:
raise ImportError(
"The `datasets` package is required for data loading. Install it with "
"`pip install fms-model-optimizer[data]`."
) from e


def quantize(
model_args: ModelArguments,
Expand Down
12 changes: 10 additions & 2 deletions fms_mo/utils/calib_data.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,11 +25,19 @@
import random

# Third Party
from datasets import load_dataset, load_from_disk
from transformers import AutoTokenizer, BatchEncoding
import datasets
import torch

try:
# Third Party
from datasets import load_dataset, load_from_disk
import datasets
except ImportError as e:
raise ImportError(
"The `datasets` package is required for data loading. Install it with "
"`pip install fms-model-optimizer[data]`."
) from e


def return_tokenized_samples(nsamples, trainenc, seqlen, sequential=False):
"""Randomly crop nsamples sequence from trainenc, each with the length of seqlen.
Expand Down
12 changes: 7 additions & 5 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -23,30 +23,32 @@ classifiers=[
dynamic = ["version"]
dependencies = [
"numpy>=1.26.4,<2.3.0",
"accelerate>=0.20.3,!=0.34,<1.11",
"transformers>4.45,<=5.12.1",
"torch>=2.2.0,<2.12.0",
"tqdm>=4.66.2,<5.0",
"datasets>=3.0.0,<5.0",
"pandas",
"safetensors",
"pkginfo>1.10"
]

[project.optional-dependencies]
examples = ["ninja>=1.11.1.1,<2.0", "evaluate", "huggingface_hub"]
# data loading for the quantization pipeline (run_quant, dq, calib_data)
data = ["datasets>=3.0.0,<5.0"]
# multi-GPU launcher (build/accelerate_launch.py) and examples
launcher = ["accelerate>=0.20.3,!=0.34,<1.11"]
examples = ["ninja>=1.11.1.1,<2.0", "evaluate", "huggingface_hub", "fms-model-optimizer[data, launcher]"]
fp8 = ["llmcompressor", "torchao==0.11"] # FP8 matmul on CPU needs a fix before advancing torchao > 0.11
fp8-infer = ["torchao==0.11"]
gptq = ["Cython", "gptqmodel>=1.7.3"]
mx = ["microxcaling>=1.1"]
opt = ["fms-model-optimizer[fp8, gptq, mx]"]
opt = ["fms-model-optimizer[fp8, gptq, mx, data, launcher]"]
aiu = ["ibm-fms>=0.0.8"]
torchvision = ["torchvision>=0.17"]
flash-attn = ["flash-attn>=2.5.3,<3.0"]
triton = ["triton>=3.0,<3.5"]
visualize = ["matplotlib", "graphviz", "pygraphviz", "tensorboard", "notebook"]
dev = ["pre-commit>=3.0.4,<5.0"]
test = ["pytest", "pillow"]
test = ["pytest", "pillow", "fms-model-optimizer[data, launcher]"]

[project.urls]
homepage = "https://github.com/foundation-model-stack/fms-model-optimizer"
Expand Down
Loading