diff --git a/lib/python3.12/site-packages/__pycache__/decorator.cpython-312.pyc b/lib/python3.12/site-packages/__pycache__/decorator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c1e63751795407e86a0ed6dec8c0f710f66a2756 Binary files /dev/null and b/lib/python3.12/site-packages/__pycache__/decorator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/__pycache__/ipython_pygments_lexers.cpython-312.pyc b/lib/python3.12/site-packages/__pycache__/ipython_pygments_lexers.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..4c147e64af329066ad6a3780eb45ea2f3ea06dd3 Binary files /dev/null and b/lib/python3.12/site-packages/__pycache__/ipython_pygments_lexers.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/__pycache__/isympy.cpython-312.pyc b/lib/python3.12/site-packages/__pycache__/isympy.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..9cf0331442847cee9a43bf8e0f2fd8eb142d3ae3 Binary files /dev/null and b/lib/python3.12/site-packages/__pycache__/isympy.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/__pycache__/pylab.cpython-312.pyc b/lib/python3.12/site-packages/__pycache__/pylab.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c96f038ed6278ca89ff8bd7e3ea2455f357b4dee Binary files /dev/null and b/lib/python3.12/site-packages/__pycache__/pylab.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/__pycache__/six.cpython-312.pyc b/lib/python3.12/site-packages/__pycache__/six.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..1b0ff50528900f54cb610038212733528c681b3b Binary files /dev/null and b/lib/python3.12/site-packages/__pycache__/six.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/click-8.2.1.dist-info/INSTALLER b/lib/python3.12/site-packages/click-8.2.1.dist-info/INSTALLER new file mode 100644 index 0000000000000000000000000000000000000000..a1b589e38a32041e49332e5e81c2d363dc418d68 --- /dev/null +++ b/lib/python3.12/site-packages/click-8.2.1.dist-info/INSTALLER @@ -0,0 +1 @@ +pip diff --git a/lib/python3.12/site-packages/click-8.2.1.dist-info/METADATA b/lib/python3.12/site-packages/click-8.2.1.dist-info/METADATA new file mode 100644 index 0000000000000000000000000000000000000000..e6c05af4a78cabdf5689f3f6f7fd57304d8a0eb1 --- /dev/null +++ b/lib/python3.12/site-packages/click-8.2.1.dist-info/METADATA @@ -0,0 +1,82 @@ +Metadata-Version: 2.4 +Name: click +Version: 8.2.1 +Summary: Composable command line interface toolkit +Maintainer-email: Pallets +Requires-Python: >=3.10 +Description-Content-Type: text/markdown +License-Expression: BSD-3-Clause +Classifier: Development Status :: 5 - Production/Stable +Classifier: Intended Audience :: Developers +Classifier: Operating System :: OS Independent +Classifier: Programming Language :: Python +Classifier: Typing :: Typed +License-File: LICENSE.txt +Requires-Dist: colorama; platform_system == 'Windows' +Project-URL: Changes, https://click.palletsprojects.com/page/changes/ +Project-URL: Chat, https://discord.gg/pallets +Project-URL: Documentation, https://click.palletsprojects.com/ +Project-URL: Donate, https://palletsprojects.com/donate +Project-URL: Source, https://github.com/pallets/click/ + +# $ click_ + +Click is a Python package for creating beautiful command line interfaces +in a composable way with as little code as necessary. It's the "Command +Line Interface Creation Kit". It's highly configurable but comes with +sensible defaults out of the box. + +It aims to make the process of writing command line tools quick and fun +while also preventing any frustration caused by the inability to +implement an intended CLI API. + +Click in three points: + +- Arbitrary nesting of commands +- Automatic help page generation +- Supports lazy loading of subcommands at runtime + + +## A Simple Example + +```python +import click + +@click.command() +@click.option("--count", default=1, help="Number of greetings.") +@click.option("--name", prompt="Your name", help="The person to greet.") +def hello(count, name): + """Simple program that greets NAME for a total of COUNT times.""" + for _ in range(count): + click.echo(f"Hello, {name}!") + +if __name__ == '__main__': + hello() +``` + +``` +$ python hello.py --count=3 +Your name: Click +Hello, Click! +Hello, Click! +Hello, Click! +``` + + +## Donate + +The Pallets organization develops and supports Click and other popular +packages. In order to grow the community of contributors and users, and +allow the maintainers to devote more time to the projects, [please +donate today][]. + +[please donate today]: https://palletsprojects.com/donate + +## Contributing + +See our [detailed contributing documentation][contrib] for many ways to +contribute, including reporting issues, requesting features, asking or answering +questions, and making PRs. + +[contrib]: https://palletsprojects.com/contributing/ + diff --git a/lib/python3.12/site-packages/click-8.2.1.dist-info/RECORD b/lib/python3.12/site-packages/click-8.2.1.dist-info/RECORD new file mode 100644 index 0000000000000000000000000000000000000000..61f4cd2b07ca70728ec4af378c37dd7e62cccddd --- /dev/null +++ b/lib/python3.12/site-packages/click-8.2.1.dist-info/RECORD @@ -0,0 +1,38 @@ +click-8.2.1.dist-info/INSTALLER,sha256=zuuue4knoyJ-UwPPXg8fezS7VCrXJQrAP7zeNuwvFQg,4 +click-8.2.1.dist-info/METADATA,sha256=dI1MbhHTLoKD2tNCCGnx9rK2gok23HDNylFeLKdLSik,2471 +click-8.2.1.dist-info/RECORD,, +click-8.2.1.dist-info/WHEEL,sha256=G2gURzTEtmeR8nrdXUJfNiB3VYVxigPQ-bEQujpNiNs,82 +click-8.2.1.dist-info/licenses/LICENSE.txt,sha256=morRBqOU6FO_4h9C9OctWSgZoigF2ZG18ydQKSkrZY0,1475 +click/__init__.py,sha256=6YyS1aeyknZ0LYweWozNZy0A9nZ_11wmYIhv3cbQrYo,4473 +click/__pycache__/__init__.cpython-312.pyc,, +click/__pycache__/_compat.cpython-312.pyc,, +click/__pycache__/_termui_impl.cpython-312.pyc,, +click/__pycache__/_textwrap.cpython-312.pyc,, +click/__pycache__/_winconsole.cpython-312.pyc,, +click/__pycache__/core.cpython-312.pyc,, +click/__pycache__/decorators.cpython-312.pyc,, +click/__pycache__/exceptions.cpython-312.pyc,, +click/__pycache__/formatting.cpython-312.pyc,, +click/__pycache__/globals.cpython-312.pyc,, +click/__pycache__/parser.cpython-312.pyc,, +click/__pycache__/shell_completion.cpython-312.pyc,, +click/__pycache__/termui.cpython-312.pyc,, +click/__pycache__/testing.cpython-312.pyc,, +click/__pycache__/types.cpython-312.pyc,, +click/__pycache__/utils.cpython-312.pyc,, +click/_compat.py,sha256=v3xBZkFbvA1BXPRkFfBJc6-pIwPI7345m-kQEnpVAs4,18693 +click/_termui_impl.py,sha256=ASXhLi9IQIc0Js9KQSS-3-SLZcPet3VqysBf9WgbbpI,26712 +click/_textwrap.py,sha256=BOae0RQ6vg3FkNgSJyOoGzG1meGMxJ_ukWVZKx_v-0o,1400 +click/_winconsole.py,sha256=_vxUuUaxwBhoR0vUWCNuHY8VUefiMdCIyU2SXPqoF-A,8465 +click/core.py,sha256=gUhpNS9cFBGdEXXdisGVG-eRvGf49RTyFagxulqwdFw,117343 +click/decorators.py,sha256=5P7abhJtAQYp_KHgjUvhMv464ERwOzrv2enNknlwHyQ,18461 +click/exceptions.py,sha256=1rdtXgHJ1b3OjGkN-UpXB9t_HCBihJvh_DtpmLmwn9s,9891 +click/formatting.py,sha256=Bhqx4QXdKQ9W4WKknIwj5KPKFmtduGOuGq1yw_THLZ8,9726 +click/globals.py,sha256=gM-Nh6A4M0HB_SgkaF5M4ncGGMDHc_flHXu9_oh4GEU,1923 +click/parser.py,sha256=nU1Ah2p11q29ul1vNdU9swPo_PUuKrxU6YXToi71q1c,18979 +click/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0 +click/shell_completion.py,sha256=CQSGdjgun4ORbOZrXP0CVhEtPx4knsufOkRsDiK64cM,19857 +click/termui.py,sha256=vAYrKC2a7f_NfEIhAThEVYfa__ib5XQbTSCGtJlABRA,30847 +click/testing.py,sha256=2eLdAaCJCGToP5Tw-XN8JjrDb3wbJIfARxg3d0crW5M,18702 +click/types.py,sha256=KBTRxN28cR1VZ5mb9iJX98MQSw_p9SGzljqfEI8z5Tw,38389 +click/utils.py,sha256=b1Mm-usEDBHtEwcPltPIn3zSK4nw2KTp5GC7_oSTlLo,20245 diff --git a/lib/python3.12/site-packages/click-8.2.1.dist-info/WHEEL b/lib/python3.12/site-packages/click-8.2.1.dist-info/WHEEL new file mode 100644 index 0000000000000000000000000000000000000000..d8b9936dad9ab2513fa6979f411560d3b6b57e37 --- /dev/null +++ b/lib/python3.12/site-packages/click-8.2.1.dist-info/WHEEL @@ -0,0 +1,4 @@ +Wheel-Version: 1.0 +Generator: flit 3.12.0 +Root-Is-Purelib: true +Tag: py3-none-any diff --git a/lib/python3.12/site-packages/click-8.2.1.dist-info/licenses/LICENSE.txt b/lib/python3.12/site-packages/click-8.2.1.dist-info/licenses/LICENSE.txt new file mode 100644 index 0000000000000000000000000000000000000000..d12a849186982399c537c5b9a8fd77bf2edd5eab --- /dev/null +++ b/lib/python3.12/site-packages/click-8.2.1.dist-info/licenses/LICENSE.txt @@ -0,0 +1,28 @@ +Copyright 2014 Pallets + +Redistribution and use in source and binary forms, with or without +modification, are permitted provided that the following conditions are +met: + +1. Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + +2. Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + +3. Neither the name of the copyright holder nor the names of its + contributors may be used to endorse or promote products derived from + this software without specific prior written permission. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS +"AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT +LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A +PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT +HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, +SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED +TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR +PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF +LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING +NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS +SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. diff --git a/lib/python3.12/site-packages/deepspeed/__init__.py b/lib/python3.12/site-packages/deepspeed/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..c7e01a0ac969325ec381d69a6a9174c3811cda52 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/__init__.py @@ -0,0 +1,398 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import sys +import types +import json +from typing import Optional, Union +import torch +from torch.optim import Optimizer +from torch.optim.lr_scheduler import _LRScheduler +from packaging import version as pkg_version + +# Skip Triton import for AMD due to pytorch-triton-rocm module breaking device API in DeepSpeed +if not (hasattr(torch.version, 'hip') and torch.version.hip is not None): + try: + import triton # noqa: F401 # type: ignore + HAS_TRITON = True + except ImportError: + HAS_TRITON = False +else: + HAS_TRITON = False + +from . import ops +from . import module_inject + +from .accelerator import get_accelerator +from .constants import TORCH_DISTRIBUTED_DEFAULT_PORT +from .runtime.engine import DeepSpeedEngine, DeepSpeedOptimizerCallable, DeepSpeedSchedulerCallable +from .runtime.engine import ADAM_OPTIMIZER, LAMB_OPTIMIZER +from .runtime.hybrid_engine import DeepSpeedHybridEngine +from .runtime.pipe.engine import PipelineEngine +from .inference.engine import InferenceEngine +from .inference.config import DeepSpeedInferenceConfig +from .runtime.lr_schedules import add_tuning_arguments +from .runtime.config import DeepSpeedConfig, DeepSpeedConfigError +from .runtime.activation_checkpointing import checkpointing +from .ops.transformer import DeepSpeedTransformerLayer, DeepSpeedTransformerConfig +from .module_inject import replace_transformer_layer, revert_transformer_layer, set_autotp_mode + +from .utils import log_dist, OnDevice, logger +from .comm.comm import init_distributed + +from .runtime import zero, domino +from .runtime.compiler import is_compile_supported + +from .pipe import PipelineModule + +from .git_version_info import version, git_hash, git_branch + + +def _parse_version(version_str): + '''Parse a version string and extract the major, minor, and patch versions.''' + ver = pkg_version.parse(version_str) + return ver.major, ver.minor, ver.micro + + +# Export version information +__version__ = version +__version_major__, __version_minor__, __version_patch__ = _parse_version(__version__) +__git_hash__ = git_hash +__git_branch__ = git_branch + +# Set to torch's distributed package or deepspeed.comm based inside DeepSpeedEngine init +dist = None + + +def initialize(args=None, + model: torch.nn.Module = None, + optimizer: Optional[Union[Optimizer, DeepSpeedOptimizerCallable]] = None, + model_parameters: Optional[torch.nn.Module] = None, + training_data: Optional[torch.utils.data.Dataset] = None, + lr_scheduler: Optional[Union[_LRScheduler, DeepSpeedSchedulerCallable]] = None, + distributed_port: int = TORCH_DISTRIBUTED_DEFAULT_PORT, + mpu=None, + dist_init_required: Optional[bool] = None, + collate_fn=None, + config=None, + mesh_param=None, + config_params=None): + """Initialize the DeepSpeed Engine. + + Arguments: + args: an object containing local_rank and deepspeed_config fields. + This is optional if `config` is passed. + + model: Required: nn.module class before apply any wrappers + + optimizer: Optional: a user defined Optimizer or Callable that returns an Optimizer object. + This overrides any optimizer definition in the DeepSpeed json config. + + model_parameters: Optional: An iterable of torch.Tensors or dicts. + Specifies what Tensors should be optimized. + + training_data: Optional: Dataset of type torch.utils.data.Dataset + + lr_scheduler: Optional: Learning Rate Scheduler Object or a Callable that takes an Optimizer and returns a Scheduler object. + The scheduler object should define a get_lr(), step(), state_dict(), and load_state_dict() methods + + distributed_port: Optional: Master node (rank 0)'s free port that needs to be used for communication during distributed training + + mpu: Optional: A model parallelism unit object that implements + get_{model,data}_parallel_{rank,group,world_size}() + + dist_init_required: Optional: None will auto-initialize torch distributed if needed, + otherwise the user can force it to be initialized or not via boolean. + + collate_fn: Optional: Merges a list of samples to form a + mini-batch of Tensor(s). Used when using batched loading from a + map-style dataset. + + config: Optional: Instead of requiring args.deepspeed_config you can pass your deepspeed config + as an argument instead, as a path or a dictionary. + + config_params: Optional: Same as `config`, kept for backwards compatibility. + + Returns: + A tuple of ``engine``, ``optimizer``, ``training_dataloader``, ``lr_scheduler`` + + * ``engine``: DeepSpeed runtime engine which wraps the client model for distributed training. + + * ``optimizer``: Wrapped optimizer if a user defined ``optimizer`` is supplied, or if + optimizer is specified in json config else ``None``. + + * ``training_dataloader``: DeepSpeed dataloader if ``training_data`` was supplied, + otherwise ``None``. + + * ``lr_scheduler``: Wrapped lr scheduler if user ``lr_scheduler`` is passed, or + if ``lr_scheduler`` specified in JSON configuration. Otherwise ``None``. + """ + log_dist("DeepSpeed info: version={}, git-hash={}, git-branch={}".format(__version__, __git_hash__, + __git_branch__), + ranks=[0]) + + # Disable zero.Init context if it's currently enabled + zero.partition_parameters.shutdown_init_context() + + assert model is not None, "deepspeed.initialize requires a model" + + global dist + from deepspeed import comm as dist + dist_backend = get_accelerator().communication_backend_name() + dist.init_distributed(dist_backend=dist_backend, + distributed_port=distributed_port, + dist_init_required=dist_init_required) + + ##TODO: combine reuse mpu as mesh device and vice versa + # Set config using config_params for backwards compat + if config is None and config_params is not None: + config = config_params + + mesh_device = None + if mesh_param: + logger.info(f"mesh_param to Initialize mesh device: {mesh_param}") + mesh_device = dist.initialize_mesh_device(mesh_param, ("data_parallel", "sequence_parallel")) + #if config file has sequence parallelize and data parallelize, then use them to initialize mesh device + elif config is not None: + if "sequence_parallel_size" in config and "data_parallel_size" in config: + logger.info(f"config to Initialize mesh device: {config}") + mesh_device = dist.initialize_mesh_device((config["data_parallel_size"], config["sequence_parallel_size"]), \ + ("data_parallel", "sequence_parallel")) + + # Check for deepscale_config for backwards compat + if hasattr(args, "deepscale_config") and args.deepscale_config is not None: + logger.warning("************ --deepscale_config is deprecated, please use --deepspeed_config ************") + if hasattr(args, "deepspeed_config"): + assert (args.deepspeed_config + is None), "Not sure how to proceed, we were given both a deepscale_config and deepspeed_config" + args.deepspeed_config = args.deepscale_config + args.deepscale_config = None + + # Check that we have only one config passed + if hasattr(args, "deepspeed_config") and args.deepspeed_config is not None: + assert config is None, "Not sure how to proceed, we were given deepspeed configs in the deepspeed arguments and deepspeed.initialize() function call" + config = args.deepspeed_config + assert config is not None, "DeepSpeed requires --deepspeed_config to specify configuration file" + if not isinstance(model, PipelineModule): + config_class = DeepSpeedConfig(config, mpu, mesh_device=mesh_device) + if config_class.hybrid_engine.enabled: + engine = DeepSpeedHybridEngine(args=args, + model=model, + optimizer=optimizer, + model_parameters=model_parameters, + training_data=training_data, + lr_scheduler=lr_scheduler, + mpu=mpu, + dist_init_required=dist_init_required, + collate_fn=collate_fn, + config=config, + config_class=config_class) + else: + engine = DeepSpeedEngine(args=args, + model=model, + optimizer=optimizer, + model_parameters=model_parameters, + training_data=training_data, + lr_scheduler=lr_scheduler, + mpu=mpu, + dist_init_required=dist_init_required, + collate_fn=collate_fn, + config=config, + mesh_device=mesh_device, + config_class=config_class) + else: + assert mpu is None, "mpu must be None with pipeline parallelism" + mpu = model.mpu() + config_class = DeepSpeedConfig(config, mpu) + engine = PipelineEngine(args=args, + model=model, + optimizer=optimizer, + model_parameters=model_parameters, + training_data=training_data, + lr_scheduler=lr_scheduler, + mpu=mpu, + dist_init_required=dist_init_required, + collate_fn=collate_fn, + config=config, + config_class=config_class) + + # Restore zero.Init context if necessary + zero.partition_parameters.restore_init_context() + + return_items = [ + engine, + engine.optimizer, + engine.training_dataloader, + engine.lr_scheduler, + ] + return tuple(return_items) + + +def _add_core_arguments(parser): + r"""Helper (internal) function to update an argument parser with an argument group of the core DeepSpeed arguments. + The core set of DeepSpeed arguments include the following: + 1) --deepspeed: boolean flag to enable DeepSpeed + 2) --deepspeed_config : path of a json configuration file to configure DeepSpeed runtime. + + This is a helper function to the public add_config_arguments() + + Arguments: + parser: argument parser + Return: + parser: Updated Parser + """ + group = parser.add_argument_group('DeepSpeed', 'DeepSpeed configurations') + + group.add_argument('--deepspeed', + default=False, + action='store_true', + help='Enable DeepSpeed (helper flag for user code, no impact on DeepSpeed backend)') + + group.add_argument('--deepspeed_config', default=None, type=str, help='DeepSpeed json configuration file.') + + group.add_argument('--deepscale', + default=False, + action='store_true', + help='Deprecated enable DeepSpeed (helper flag for user code, no impact on DeepSpeed backend)') + + group.add_argument('--deepscale_config', + default=None, + type=str, + help='Deprecated DeepSpeed json configuration file.') + + return parser + + +def add_config_arguments(parser): + r"""Update the argument parser to enabling parsing of DeepSpeed command line arguments. + The set of DeepSpeed arguments include the following: + 1) --deepspeed: boolean flag to enable DeepSpeed + 2) --deepspeed_config : path of a json configuration file to configure DeepSpeed runtime. + + Arguments: + parser: argument parser + Return: + parser: Updated Parser + """ + parser = _add_core_arguments(parser) + + return parser + + +def default_inference_config(): + """ + Return a default DeepSpeed inference configuration dictionary. + """ + return DeepSpeedInferenceConfig().dict() + + +def init_inference(model, config=None, **kwargs): + """Initialize the DeepSpeed InferenceEngine. + + Description: all four cases are valid and supported in DS init_inference() API. + + # Case 1: user provides no config and no kwargs. Default config will be used. + + .. code-block:: python + + generator.model = deepspeed.init_inference(generator.model) + string = generator("DeepSpeed is") + print(string) + + # Case 2: user provides a config and no kwargs. User supplied config will be used. + + .. code-block:: python + + generator.model = deepspeed.init_inference(generator.model, config=config) + string = generator("DeepSpeed is") + print(string) + + # Case 3: user provides no config and uses keyword arguments (kwargs) only. + + .. code-block:: python + + generator.model = deepspeed.init_inference(generator.model, + tensor_parallel={"tp_size": world_size}, + dtype=torch.half, + replace_with_kernel_inject=True) + string = generator("DeepSpeed is") + print(string) + + # Case 4: user provides config and keyword arguments (kwargs). Both config and kwargs are merged and kwargs take precedence. + + .. code-block:: python + + generator.model = deepspeed.init_inference(generator.model, config={"dtype": torch.half}, replace_with_kernel_inject=True) + string = generator("DeepSpeed is") + print(string) + + Arguments: + model: Required: original nn.module object without any wrappers + + config: Optional: instead of arguments, you can pass in a DS inference config dict or path to JSON file + + Returns: + A deepspeed.InferenceEngine wrapped model. + """ + log_dist("DeepSpeed info: version={}, git-hash={}, git-branch={}".format(__version__, __git_hash__, + __git_branch__), + ranks=[0]) + + # Load config_dict from config first + if config is None: + config = {} + if isinstance(config, str): + with open(config, "r") as f: + config_dict = json.load(f) + elif isinstance(config, dict): + config_dict = config + else: + raise ValueError(f"'config' argument expected string or dictionary, got {type(config)}") + + # Update with values from kwargs, ensuring no conflicting overlap between config and kwargs + overlap_keys = set(config_dict.keys()).intersection(kwargs.keys()) + # If there is overlap, error out if values are different + for key in overlap_keys: + if config_dict[key] != kwargs[key]: + raise ValueError(f"Conflicting argument '{key}' in 'config':{config_dict[key]} and kwargs:{kwargs[key]}") + config_dict.update(kwargs) + + ds_inference_config = DeepSpeedInferenceConfig(**config_dict) + + engine = InferenceEngine(model, config=ds_inference_config) + + return engine + + +def tp_model_init(model, tp_size, dtype, config=None, **kwargs): + """ + Initialize the model for tensor parallelism. + + Args: + model (torch.nn.Module): The model to be initialized. + tp_size (int): The tensor parallelism size. + dtype (torch.dtype): The data type to be used for the model. + + Returns: + torch.nn.Module: The initialized model with tensor parallelism. + """ + # avoid re-entry + if hasattr(model, 'ds_autotp_parsed'): + logger.warning("ds_autotp_parsed' attribute already exists in the model, re-entry is not allowed.") + return + + set_autotp_mode(training=True) + + from deepspeed.runtime.tensor_parallel import TpTrainingManager + # The expected usage here is for it to be invoked by transformers package. + + #TODO: We should provide a custom TP mapping solution without using autoTP + #as modifying the autoTP logic may be more difficult for users compared to configuring it + + model = TpTrainingManager(model=model, tp_size=tp_size, dtype=dtype).module + + setattr(model, 'ds_autotp_parsed', True) + + return model diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__init__.py b/lib/python3.12/site-packages/deepspeed/accelerator/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..efed1ef84aca357f5a2f1a06d41c254c01f36a3e --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/accelerator/__init__.py @@ -0,0 +1,7 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .abstract_accelerator import DeepSpeedAccelerator +from .real_accelerator import get_accelerator, set_accelerator, is_current_accelerator_supported diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..56f23c40bdf5246a54102ce2a950aadb8916e1e3 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/abstract_accelerator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/abstract_accelerator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..181589dcace623f706dc53350f0f803addb11255 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/abstract_accelerator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/cpu_accelerator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/cpu_accelerator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..0e9994df2ebbd6ca6e7c0eb1d9346841b7768fb5 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/cpu_accelerator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/cuda_accelerator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/cuda_accelerator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..8327745402131743d2bc8de41449e5f19295b4bd Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/cuda_accelerator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/hpu_accelerator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/hpu_accelerator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7b6e9992a9c8445e8e1d30ebf8f349e43acebcfb Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/hpu_accelerator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/mlu_accelerator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/mlu_accelerator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..45a631a8807108551199820c4f3d799df487d8af Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/mlu_accelerator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/mps_accelerator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/mps_accelerator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..33eb1cfc993374f4ef5e5d5ffa9f0f47d7ed71aa Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/mps_accelerator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/npu_accelerator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/npu_accelerator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..04fe8bd164dd96f3111e37b7a1dae74821c69c7a Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/npu_accelerator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/real_accelerator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/real_accelerator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..74a9cbf51ec6f7c370e39bd35c05978d5d7c313b Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/real_accelerator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/sdaa_accelerator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/sdaa_accelerator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c22260e85f570f9c2665bb561a0ca56b14d4f081 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/sdaa_accelerator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/xpu_accelerator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/xpu_accelerator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..5dd34e7b2923bdc8c117cb83b2ef736a1a77b189 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/accelerator/__pycache__/xpu_accelerator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/abstract_accelerator.py b/lib/python3.12/site-packages/deepspeed/accelerator/abstract_accelerator.py new file mode 100644 index 0000000000000000000000000000000000000000..2a0770ac681b7236f5f121c179b921fab9dd8a9e --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/accelerator/abstract_accelerator.py @@ -0,0 +1,306 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import abc +from abc import ABC + + +class DeepSpeedAccelerator(ABC): + + def __init__(self): + self._name = None + self._communication_backend_name = None + self._compile_backend = None + + @abc.abstractmethod + def is_synchronized_device(self): + ... + + @abc.abstractmethod + def use_host_timers(self): + ... + + @abc.abstractmethod + def resolves_data_dependency(self): + ... + + @abc.abstractmethod + def handles_memory_backpressure(self): + ... + + # Device APIs + @abc.abstractmethod + def device_name(self, device_index): + ... + + @abc.abstractmethod + def device(self, device_index): + ... + + @abc.abstractmethod + def set_device(self, device_index): + ... + + @abc.abstractmethod + def current_device(self): + ... + + @abc.abstractmethod + def current_device_name(self): + ... + + @abc.abstractmethod + def device_count(self): + ... + + @abc.abstractmethod + def synchronize(self, device_index=None): + ... + + # RNG APIs + @abc.abstractmethod + def random(self): + ... + + @abc.abstractmethod + def set_rng_state(self, new_state, device_index=None): + ... + + @abc.abstractmethod + def get_rng_state(self, device_index=None): + ... + + @abc.abstractmethod + def manual_seed(self, seed): + ... + + @abc.abstractmethod + def manual_seed_all(self, seed): + ... + + @abc.abstractmethod + def initial_seed(self): + ... + + @abc.abstractmethod + def default_generator(self, device_index): + ... + + # Streams/Events + @property + @abc.abstractmethod + def Stream(self): + ... + + @abc.abstractmethod + def stream(self, stream): + ... + + @abc.abstractmethod + def current_stream(self, device_index=None): + ... + + @abc.abstractmethod + def default_stream(self, device_index=None): + ... + + @property + @abc.abstractmethod + def Event(self): + ... + + # Memory management + @abc.abstractmethod + def empty_cache(self): + ... + + @abc.abstractmethod + def memory_allocated(self, device_index=None): + ... + + @abc.abstractmethod + def max_memory_allocated(self, device_index=None): + ... + + @abc.abstractmethod + def reset_max_memory_allocated(self, device_index=None): + ... + + @abc.abstractmethod + def memory_cached(self, device_index=None): + ... + + @abc.abstractmethod + def max_memory_cached(self, device_index=None): + ... + + @abc.abstractmethod + def reset_max_memory_cached(self, device_index=None): + ... + + @abc.abstractmethod + def memory_stats(self, device_index=None): + ... + + @abc.abstractmethod + def reset_peak_memory_stats(self, device_index=None): + ... + + @abc.abstractmethod + def memory_reserved(self, device_index=None): + ... + + @abc.abstractmethod + def max_memory_reserved(self, device_index=None): + ... + + @abc.abstractmethod + def total_memory(self, device_index=None): + ... + + @abc.abstractmethod + def available_memory(self, device_index=None): + ... + + # Data types + @abc.abstractmethod + def is_bf16_supported(self): + ... + + @abc.abstractmethod + def is_fp16_supported(self): + ... + + @abc.abstractmethod + def supported_dtypes(self): + ... + + # Misc + @abc.abstractmethod + def amp(self): + ... + + @abc.abstractmethod + def is_available(self): + ... + + @abc.abstractmethod + def range_push(self, msg): + ... + + @abc.abstractmethod + def range_pop(self): + ... + + @abc.abstractmethod + def lazy_call(self, callback): + ... + + @abc.abstractmethod + def communication_backend_name(self): + ... + + @abc.abstractmethod + def is_triton_supported(self): + ... + + # Graph operations + @abc.abstractmethod + def create_graph(self): + ... + + @abc.abstractmethod + def capture_to_graph(self, graph, pool=None, stream=None): + ... + + @abc.abstractmethod + def replay_graph(self, graph): + ... + + # Tensor operations + @property + @abc.abstractmethod + def BFloat16Tensor(self): + ... + + @property + @abc.abstractmethod + def ByteTensor(self): + ... + + @property + @abc.abstractmethod + def DoubleTensor(self): + ... + + @property + @abc.abstractmethod + def FloatTensor(self): + ... + + @property + @abc.abstractmethod + def HalfTensor(self): + ... + + @property + @abc.abstractmethod + def IntTensor(self): + ... + + @property + @abc.abstractmethod + def LongTensor(self): + ... + + @abc.abstractmethod + def pin_memory(self, tensor, align_bytes=1): + ... + + @abc.abstractmethod + def is_pinned(self, tensor): + ... + + @abc.abstractmethod + def on_accelerator(self, tensor): + ... + + @abc.abstractmethod + def op_builder_dir(self): + ... + + # create an instance of op builder, specified by class_name + @abc.abstractmethod + def create_op_builder(self, class_name): + ... + + # return an op builder class, specified by class_name + @abc.abstractmethod + def get_op_builder(self, class_name): + ... + + @abc.abstractmethod + def build_extension(self): + ... + + @abc.abstractmethod + def export_envs(self): + ... + + @abc.abstractmethod + def visible_devices_envs(self): + ... + + @abc.abstractmethod + def set_visible_devices_envs(self, current_env, local_accelerator_ids): + ... + + @abc.abstractmethod + def get_compile_backend(self): + ... + + @abc.abstractmethod + def set_compile_backend(self, backend): + ... diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/cpu_accelerator.py b/lib/python3.12/site-packages/deepspeed/accelerator/cpu_accelerator.py new file mode 100644 index 0000000000000000000000000000000000000000..4b3d89e6cd34e42514cd27e75d5a9a3a974d96cf --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/accelerator/cpu_accelerator.py @@ -0,0 +1,361 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .abstract_accelerator import DeepSpeedAccelerator + +# During setup stage torch may not be installed, pass on no torch will +# allow op builder related API to be executed. +try: + import torch +except ImportError as e: + pass + +try: + import oneccl_bindings_for_pytorch # noqa: F401 # type: ignore + oneccl_imported_p = True +except ImportError as e: + oneccl_imported_p = False + +import os + + +# accelerator for Intel CPU +class CPU_Accelerator(DeepSpeedAccelerator): + + def __init__(self): + self._name = 'cpu' + self._compile_backend = "inductor" + if oneccl_imported_p: + self._communication_backend_name = 'ccl' + else: + # fallback to gloo if oneccl_binding_for_pytorch is not installed + self._communication_backend_name = 'gloo' + try: + import psutil + mem = psutil.Process().memory_info().rss + self.max_mem = mem + except ImportError as e: + self.max_mem = 0 + + def is_synchronized_device(self): + return True + + def use_host_timers(self): + return self.is_synchronized_device() + + def resolves_data_dependency(self): + return self.is_synchronized_device() + + def handles_memory_backpressure(self): + return self.is_synchronized_device() + + # Device APIs + def device_name(self, device_index=None): + return 'cpu' + + def device(self, device_index=None): + return None + + def set_device(self, device_index): + return + + def current_device(self): + return os.environ.get('LOCAL_RANK', 0) + + def current_device_name(self): + return 'cpu' + + def device_count(self): + device_count = int(os.environ.get('LOCAL_SIZE', 0)) + if device_count > 0: + return device_count + else: + from deepspeed.utils.numa import get_numa_cores + # Count NUMA node for number of cpu accelerators. On machine with HBM + # In flat mode, HBM is in separate NUMA node with no cores on this node. + # Ignore these NUMA nodes with no cores. + numa_core_lists = get_numa_cores() + if not numa_core_lists: + return 1 + numa_count = 0 + prev_core_list = [] + for core_list in numa_core_lists: + if len(core_list) > 0 and core_list != prev_core_list: + numa_count += 1 + prev_core_list = core_list + return numa_count + + def synchronize(self, device_index=None): + return + + # RNG APIs + def random(self): + return torch.random + + def set_rng_state(self, new_state, device_index=None): + if device_index is None: + return torch.set_rng_state(new_state) + return torch.set_rng_state(new_state, device_index) + + def get_rng_state(self, device_index=None): + return torch.get_rng_state() + + def manual_seed(self, seed): + return torch.manual_seed(seed) + + def manual_seed_all(self, seed): + return torch.manual_seed(seed) + + def initial_seed(self): + return torch.initial_seed() + + def default_generator(self, device_index): + return torch.default_generator + + # Streams/Events + @property + def Stream(self): + return None + + def stream(self, stream): + from deepspeed.runtime.utils import noop_context + return noop_context() + + def current_stream(self, device_index=None): + return None + + def default_stream(self, device_index=None): + return None + + @property + def Event(self): + return None + + # Memory management + def empty_cache(self): + return + + def get_rss(self): + import psutil + mem = psutil.Process().memory_info().rss + if mem > self.max_mem: + self.max_mem = mem + return mem + + def reset_rss(self): + import psutil + mem = psutil.Process().memory_info().rss + self.max_mem = mem + return mem + + def memory_allocated(self, device_index=None): + return self.get_rss() + + def max_memory_allocated(self, device_index=None): + self.get_rss() + return self.max_mem + + def reset_max_memory_allocated(self, device_index=None): + self.reset_rss() + return + + def memory_cached(self, device_index=None): + return self.get_rss() + + def max_memory_cached(self, device_index=None): + self.get_rss() + return self.max_mem + + def reset_max_memory_cached(self, device_index=None): + self.reset_rss() + return + + def memory_stats(self, device_index=None): + mem = self.get_rss() + mem_stat = {} + mem_stat['allocated_bytes.all.current'] = mem + mem_stat['allocated_bytes.all.peak'] = self.max_mem + return mem_stat + + def reset_peak_memory_stats(self, device_index=None): + self.reset_rss() + return + + def memory_reserved(self, device_index=None): + return self.get_rss() + + def max_memory_reserved(self, device_index=None): + self.get_rss() + return self.max_mem + + def total_memory(self, device_index=None): + import psutil + return psutil.virtual_memory().total + + def available_memory(self, device_index=None): + import psutil + return psutil.virtual_memory().available + + # Misc + def amp(self): + return torch.cpu.amp + + def is_available(self): + return True + + def range_push(self, msg): + # TODO itt is currently not supported yet + # return torch.profiler.itt.range_push(msg) + return + + def range_pop(self): + # TODO itt is currently not supported yet + # return torch.profiler.itt.range_pop() + return + + def lazy_call(self, callback): + return callback() + + def communication_backend_name(self): + return self._communication_backend_name + + def is_triton_supported(self): + return False + + # Data types + def is_bf16_supported(self): + return True + + def is_fp16_supported(self): + try: + if torch.ops.mkldnn._is_mkldnn_fp16_supported(): + return True + except: + return False + + def supported_dtypes(self): + supported_dtypes = [torch.float, torch.bfloat16] + if self.is_fp16_supported(): + supported_dtypes.append(torch.float16) + return supported_dtypes + + # Graph operations + def create_graph(self): + return None + + def capture_to_graph(self, graph, pool=None, stream=None): + from deepspeed.runtime.utils import noop_context + return noop_context() + + def replay_graph(self, graph): + return + + # Tensor operations + @property + def BFloat16Tensor(self): + return torch.BFloat16Tensor + + @property + def ByteTensor(self): + return torch.ByteTensor + + @property + def DoubleTensor(self): + return torch.DoubleTensor + + @property + def FloatTensor(self): + return torch.FloatTensor + + @property + def HalfTensor(self): + return torch.HalfTensor + + @property + def IntTensor(self): + return torch.IntTensor + + @property + def LongTensor(self): + return torch.LongTensor + + def pin_memory(self, tensor, align_bytes=1): + return tensor + + def is_pinned(self, tensor): + return tensor.is_pinned() + + def op_builder_dir(self): + try: + # is op_builder from deepspeed or a 3p version? this should only succeed if it's deepspeed + # if successful this also means we're doing a local install and not JIT compile path + from op_builder import __deepspeed__ # noqa: F401 # type: ignore + return "op_builder.cpu" + except ImportError: + return "deepspeed.ops.op_builder.cpu" + + def on_accelerator(self, tensor): + device_str = str(tensor.device) + if device_str.startswith('cpu'): + return True + else: + return False + + # create an instance of op builder and return, name specified by class_name + def create_op_builder(self, op_name): + builder_class = self.get_op_builder(op_name) + if builder_class is not None: + return builder_class() + return None + + # return an op builder class, name specified by class_name + def get_op_builder(self, class_name): + try: + # is op_builder from deepspeed or a 3p version? this should only succeed if it's deepspeed + # if successful this also means we're doing a local install and not JIT compile path + from op_builder import __deepspeed__ # noqa: F401 # type: ignore + from op_builder.cpu import AsyncIOBuilder, CCLCommBuilder, ShareMemCommBuilder, FusedAdamBuilder, CPUAdamBuilder, NotImplementedBuilder + except ImportError: + from deepspeed.ops.op_builder.cpu import AsyncIOBuilder, CCLCommBuilder, ShareMemCommBuilder, FusedAdamBuilder, CPUAdamBuilder, NotImplementedBuilder + + if class_name == "CCLCommBuilder": + return CCLCommBuilder + elif class_name == "ShareMemCommBuilder": + return ShareMemCommBuilder + elif class_name == "FusedAdamBuilder": + return FusedAdamBuilder + elif class_name == "CPUAdamBuilder": + return CPUAdamBuilder + elif class_name == "AsyncIOBuilder": + return AsyncIOBuilder + else: + # return a NotImplementedBuilder to avoid get NoneType[Name] in unit tests + return NotImplementedBuilder + + def build_extension(self): + from torch.utils.cpp_extension import BuildExtension + return BuildExtension + + def export_envs(self): + return [] + + # TODO: cpu's visible envs is confirmed, keep as CUDA_VISIBLE_DEVICES + def visible_devices_envs(self): + return ['CUDA_VISIBLE_DEVICES'] + + def set_visible_devices_envs(self, current_env, local_accelerator_ids): + for env in self.visible_devices_envs(): + current_env[env] = ",".join(map(str, local_accelerator_ids)) + + def get_compile_backend(self): + return self._compile_backend + + def set_compile_backend(self, backend): + supported_backends = torch._dynamo.list_backends(exclude_tags=()) + if backend in supported_backends: + self._compile_backend = backend + else: + raise ValueError( + f"{backend} not supported by {self.device_name()}. Supported Backends are {supported_backends}") diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/cuda_accelerator.py b/lib/python3.12/site-packages/deepspeed/accelerator/cuda_accelerator.py new file mode 100644 index 0000000000000000000000000000000000000000..06fd443f98295fcd9f434d2a4016bc9b8416bfc0 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/accelerator/cuda_accelerator.py @@ -0,0 +1,382 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import functools +import os +import pkgutil +import importlib +import sys + +from .abstract_accelerator import DeepSpeedAccelerator +# During setup stage torch may not be installed, pass on no torch will +# allow op builder related API to be executed. +try: + import torch.cuda +except ImportError: + pass + +# Delay import pynvml to avoid import error when CUDA is not available +pynvml = None + + +class CUDA_Accelerator(DeepSpeedAccelerator): + + def __init__(self): + self._name = 'cuda' + self._communication_backend_name = 'nccl' if sys.platform != 'win32' else 'gloo' + self._compile_backend = "inductor" + if pynvml is None: + self._init_pynvml() + + def _init_pynvml(self): + global pynvml + try: + import pynvml + except ImportError: + return + try: + pynvml.nvmlInit() + except pynvml.NVMLError: + pynvml = None + return + + def is_synchronized_device(self): + return False + + def use_host_timers(self): + return self.is_synchronized_device() + + def resolves_data_dependency(self): + return self.is_synchronized_device() + + def handles_memory_backpressure(self): + return self.is_synchronized_device() + + # Device APIs + def device_name(self, device_index=None): + if device_index is None: + return 'cuda' + return 'cuda:{}'.format(device_index) + + def device(self, device_index=None): + return torch.cuda.device(device_index) + + def set_device(self, device_index): + torch.cuda.set_device(device_index) + + def current_device(self): + return torch.cuda.current_device() + + def current_device_name(self): + return 'cuda:{}'.format(torch.cuda.current_device()) + + def device_count(self): + return torch.cuda.device_count() + + def synchronize(self, device_index=None): + return torch.cuda.synchronize(device_index) + + # RNG APIs + def random(self): + return torch.random + + def set_rng_state(self, new_state, device_index=None): + if device_index is None: + return torch.cuda.set_rng_state(new_state) + + return torch.cuda.set_rng_state(new_state, device_index) + + def get_rng_state(self, device_index=None): + if device_index is None: + return torch.cuda.get_rng_state() + + return torch.cuda.get_rng_state(device_index) + + def manual_seed(self, seed): + return torch.cuda.manual_seed(seed) + + def manual_seed_all(self, seed): + return torch.cuda.manual_seed_all(seed) + + def initial_seed(self): + return torch.cuda.initial_seed() + + def default_generator(self, device_index): + return torch.cuda.default_generators[device_index] + + # Streams/Events + @property + def Stream(self): + return torch.cuda.Stream + + def stream(self, stream): + return torch.cuda.stream(stream) + + def current_stream(self, device_index=None): + return torch.cuda.current_stream(device_index) + + def default_stream(self, device_index=None): + return torch.cuda.default_stream(device_index) + + @property + def Event(self): + return torch.cuda.Event + + # Memory management + def empty_cache(self): + return torch.cuda.empty_cache() + + def memory_allocated(self, device_index=None): + return torch.cuda.memory_allocated(device_index) + + def max_memory_allocated(self, device_index=None): + return torch.cuda.max_memory_allocated(device_index) + + def reset_max_memory_allocated(self, device_index=None): + return torch.cuda.reset_max_memory_allocated(device_index) + + def memory_cached(self, device_index=None): + return torch.cuda.memory_cached(device_index) + + def max_memory_cached(self, device_index=None): + return torch.cuda.max_memory_cached(device_index) + + def reset_max_memory_cached(self, device_index=None): + return torch.cuda.reset_max_memory_cached(device_index) + + def memory_stats(self, device_index=None): + if hasattr(torch.cuda, 'memory_stats'): + return torch.cuda.memory_stats(device_index) + + def reset_peak_memory_stats(self, device_index=None): + if hasattr(torch.cuda, 'reset_peak_memory_stats'): + return torch.cuda.reset_peak_memory_stats(device_index) + + def memory_reserved(self, device_index=None): + if hasattr(torch.cuda, 'memory_reserved'): + return torch.cuda.memory_reserved(device_index) + + def max_memory_reserved(self, device_index=None): + if hasattr(torch.cuda, 'max_memory_reserved'): + return torch.cuda.max_memory_reserved(device_index) + + def total_memory(self, device_index=None): + return torch.cuda.get_device_properties(device_index).total_memory + + def _get_nvml_gpu_id(self, torch_gpu_id): + """ + credit: https://discuss.pytorch.org/t/making-pynvml-match-torch-device-ids-cuda-visible-devices/103020 + + Remap torch device id to nvml device id, respecting CUDA_VISIBLE_DEVICES. + + If the latter isn't set return the same id + """ + # if CUDA_VISIBLE_DEVICES is used automagically remap the id since pynvml ignores this env var + if "CUDA_VISIBLE_DEVICES" in os.environ: + ids = list(map(int, os.environ.get("CUDA_VISIBLE_DEVICES", "").split(","))) + return ids[torch_gpu_id] # remap + else: + return torch_gpu_id + + def available_memory(self, device_index=None): + if pynvml: + if device_index is None: + device_index = self.current_device() + handle = pynvml.nvmlDeviceGetHandleByIndex(self._get_nvml_gpu_id(device_index)) + info = pynvml.nvmlDeviceGetMemoryInfo(handle) + return info.free + else: + return self.total_memory(device_index) - self.memory_allocated(device_index) + + # Data types + def is_bf16_supported(self): + if not torch.cuda.is_available(): + return True + return torch.cuda.is_bf16_supported() + + def is_fp16_supported(self): + if not torch.cuda.is_available(): + return True + # See https://docs.nvidia.com/deeplearning/tensorrt/support-matrix/index.html#hardware-precision-matrix + # FP16 on compute capability 6.x is deprecated + allow_deprecated_fp16 = os.environ.get('DS_ALLOW_DEPRECATED_FP16', '0') == '1' + major, _ = torch.cuda.get_device_capability() + if major >= 7: + return True + elif major == 6 and allow_deprecated_fp16: + return True + else: + return False + + def supported_dtypes(self): + supported_dtypes = [torch.float] + if self.is_fp16_supported(): + supported_dtypes.append(torch.half) + if self.is_bf16_supported(): + supported_dtypes.append(torch.bfloat16) + return supported_dtypes + + # Misc + def amp(self): + if hasattr(torch.cuda, 'amp'): + return torch.cuda.amp + return None + + def is_available(self): + return torch.cuda.is_available() + + def range_push(self, msg): + if hasattr(torch.cuda.nvtx, 'range_push'): + return torch.cuda.nvtx.range_push(msg) + + def range_pop(self): + if hasattr(torch.cuda.nvtx, 'range_pop'): + return torch.cuda.nvtx.range_pop() + + def lazy_call(self, callback): + return torch.cuda._lazy_call(callback) + + def communication_backend_name(self): + return self._communication_backend_name + + def is_triton_supported(self): + major, _ = torch.cuda.get_device_capability() + if major >= 8: + return True + else: + return False + + # Graph operations + def create_graph(self): + return torch.cuda.CUDAGraph() + + def capture_to_graph(self, graph, pool=None, stream=None): + return torch.cuda.graph(graph, pool, stream) + + def replay_graph(self, graph): + graph.replay() + return + + # Tensor operations + + @property + def BFloat16Tensor(self): + return functools.partial(torch.tensor, dtype=torch.bfloat16, device='cuda') + + @property + def ByteTensor(self): + return functools.partial(torch.tensor, dtype=torch.uint8, device='cuda') + + @property + def DoubleTensor(self): + return functools.partial(torch.tensor, dtype=torch.double, device='cuda') + + @property + def FloatTensor(self): + return functools.partial(torch.tensor, dtype=torch.float, device='cuda') + + @property + def HalfTensor(self): + return functools.partial(torch.tensor, dtype=torch.half, device='cuda') + + @property + def IntTensor(self): + return functools.partial(torch.tensor, dtype=torch.int, device='cuda') + + @property + def LongTensor(self): + return functools.partial(torch.tensor, dtype=torch.long, device='cuda') + + def pin_memory(self, tensor, align_bytes=1): + return tensor.pin_memory() + + def is_pinned(self, tensor): + return tensor.is_pinned() + + def on_accelerator(self, tensor): + device_str = str(tensor.device) + if device_str.startswith('cuda:'): + return True + else: + return False + + def op_builder_dir(self): + try: + # is op_builder from deepspeed or a 3p version? this should only succeed if it's deepspeed + # if successful this also means we're doing a local install and not JIT compile path + from op_builder import __deepspeed__ # noqa: F401 # type: ignore + return "op_builder" + except ImportError: + return "deepspeed.ops.op_builder" + + # dict that holds class name <--> class type mapping i.e. + # 'AsyncIOBuilder': + # this dict will be filled at init stage + class_dict = None + + def _lazy_init_class_dict(self): + if self.class_dict is not None: + return + else: + self.class_dict = {} + # begin initialize for create_op_builder() + # put all valid class name <--> class type mapping into class_dict + op_builder_dir = self.op_builder_dir() + op_builder_module = importlib.import_module(op_builder_dir) + op_builder_absolute_path = os.path.dirname(op_builder_module.__file__) + for _, module_name, _ in pkgutil.iter_modules([op_builder_absolute_path]): + # avoid self references, + # skip sub_directories which contains ops for other backend(cpu, npu, etc.). + if module_name != 'all_ops' and module_name != 'builder' and not os.path.isdir( + os.path.join(op_builder_absolute_path, module_name)): + module = importlib.import_module("{}.{}".format(op_builder_dir, module_name)) + for member_name in module.__dir__(): + if member_name.endswith( + 'Builder' + ) and member_name != "OpBuilder" and member_name != "CUDAOpBuilder" and member_name != "TorchCPUOpBuilder": # avoid abstract classes + if not member_name in self.class_dict: + self.class_dict[member_name] = getattr(module, member_name) + # end initialize for create_op_builder() + + # create an instance of op builder and return, name specified by class_name + def create_op_builder(self, class_name): + self._lazy_init_class_dict() + if class_name in self.class_dict: + return self.class_dict[class_name]() + else: + return None + + # return an op builder class, name specified by class_name + def get_op_builder(self, class_name): + self._lazy_init_class_dict() + if class_name in self.class_dict: + return self.class_dict[class_name] + else: + return None + + def build_extension(self): + from torch.utils.cpp_extension import BuildExtension + return BuildExtension + + def export_envs(self): + return ['NCCL'] + + def visible_devices_envs(self): + return ['CUDA_VISIBLE_DEVICES'] + + def set_visible_devices_envs(self, current_env, local_accelerator_ids): + for env in self.visible_devices_envs(): + current_env[env] = ",".join(map(str, local_accelerator_ids)) + + def get_compile_backend(self): + return self._compile_backend + + def set_compile_backend(self, backend): + supported_backends = torch._dynamo.list_backends(exclude_tags=()) + if backend in supported_backends: + self._compile_backend = backend + else: + raise ValueError( + f"{backend} not supported by {self.device_name()}. Supported Backends are {supported_backends}") diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/hpu_accelerator.py b/lib/python3.12/site-packages/deepspeed/accelerator/hpu_accelerator.py new file mode 100644 index 0000000000000000000000000000000000000000..cae1fa0383cb58d64a051aeb265b2da8e5ed5ed4 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/accelerator/hpu_accelerator.py @@ -0,0 +1,331 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import functools +import os +import pkgutil +import importlib +import torch + +from .abstract_accelerator import DeepSpeedAccelerator + + +class HPU_Accelerator(DeepSpeedAccelerator): + + def __init__(self): + self._name = 'hpu' + self._communication_backend_name = 'hccl' + self._compile_backend = "hpu_backend" + self.apply_hpu_workarounds() + try: + import habana_frameworks.torch.hpu as hpu + self.hpu = hpu + torch.use_deterministic_algorithms(True) + # TODO: remove this WA when memory mapping break is resolved. + torch.utils.deterministic.fill_uninitialized_memory = False + except ImportError as e: + raise ValueError( + f"HPU_Accelerator requires habana_frameworks.torch.hpu, which is not installed on this system.") + + self.fp16_supported = None + + def apply_hpu_workarounds(self): + + def update_wa_env_var(key, value): + if key not in os.environ.keys(): + os.environ[key] = value + + update_wa_env_var("PT_HPU_LAZY_ACC_PAR_MODE", "0") + update_wa_env_var("PT_HPU_ENABLE_REFINE_DYNAMIC_SHAPES", "0") + + # Device APIs + def is_synchronized_device(self): + return False + + def use_host_timers(self): + return False + + def resolves_data_dependency(self): + return True + + def handles_memory_backpressure(self): + return True + + def device_name(self, device_index=None): + # ignoring device_index. + return 'hpu' + + def device(self, device_index=None): + return torch.device(self.device_name(device_index)) + + def set_device(self, device_index): + self.hpu.set_device(device_index) + + def current_device(self): + return (self.hpu.current_device()) + + def current_device_name(self): + return 'hpu:{}'.format(self.current_device()) + + def device_count(self): + return self.hpu.device_count() + + def synchronize(self, device_index=None): + return self.hpu.synchronize() + + # RNG APIs + def random(self): + return torch.random + + def set_rng_state(self, new_state, device_index=None): + self.hpu.random.set_rng_state(new_state) + + def get_rng_state(self, device_index=None): + return self.hpu.random.get_rng_state() + + def manual_seed(self, seed): + return self.hpu.random.manual_seed(seed) + + def manual_seed_all(self, seed): + self.hpu.random.manual_seed_all(seed) + + def initial_seed(self): + return self.hpu.random.initial_seed() + + def default_generator(self, device_index): + return self.hpu.random.default_generators[device_index] + + # Streams/Events + @property + def Stream(self): + return self.hpu.Stream + + def stream(self, stream): + return self.hpu.stream(stream) + + def current_stream(self, device_index=None): + return self.hpu.current_stream() + + def default_stream(self, device_index=None): + return self.hpu.default_stream() + + @property + def Event(self): + import habana_frameworks.torch.core as htcore + return htcore.hpu.Event + + # Memory management + def empty_cache(self): + return + + def memory_allocated(self, device_index=None): + return self.hpu.memory_allocated() + + def max_memory_allocated(self, device_index=None): + return self.hpu.max_memory_allocated() + + def reset_max_memory_allocated(self, device_index=None): + return self.hpu.reset_max_memory_allocated() + + def memory_cached(self, device_index=None): + return self.hpu.memory_cached(device_index) + + def max_memory_cached(self, device_index=None): + return self.hpu.max_memory_cached(device_index) + + def reset_max_memory_cached(self, device_index=None): + return None + + def memory_stats(self, device_index=None): + return self.hpu.memory_stats(device_index) + + def reset_peak_memory_stats(self, device_index=None): + self.hpu.reset_peak_memory_stats(device_index) + + def memory_reserved(self, device_index=None): + return self.hpu.memory_reserved(device_index) + + def max_memory_reserved(self, device_index=None): + return self.hpu.max_memory_reserved(device_index) + + def total_memory(self, device_index=None): + return self.memory_stats(device_index)['Limit'] + + def available_memory(self, device_index=None): + return self.total_memory(device_index) - self.memory_allocated(device_index) + + # Data types + def is_bf16_supported(self): + return True + + def is_fp16_supported(self): + if self.fp16_supported is None: + import habana_frameworks.torch.utils.experimental as htexp + self.fp16_supported = htexp._is_fp16_supported() + return self.fp16_supported + + def supported_dtypes(self): + supported_dtypes = [torch.float, torch.bfloat16] + if self.is_fp16_supported(): + supported_dtypes.append(torch.half) + return supported_dtypes + + # Misc + def amp(self): + return None + + def is_available(self): + return self.hpu.is_available() + + def range_push(self, msg): + return + + def range_pop(self): + return + + def lazy_call(self, callback): + callback() + + def communication_backend_name(self): + return self._communication_backend_name + + def is_triton_supported(self): + return False + + # Graph operations + def create_graph(self): + return self.hpu.HPUGraph() + + def capture_to_graph(self, graph, pool=None, stream=None): + return self.hpu.graph(graph, stream=stream) + + def replay_graph(self, graph): + graph.replay() + return + + # Tensor operations + @property + def BFloat16Tensor(self): + return functools.partial(torch.tensor, dtype=torch.bfloat16, device='hpu') + + @property + def ByteTensor(self): + return functools.partial(torch.tensor, dtype=torch.uint8, device='hpu') + + @property + def DoubleTensor(self): + return functools.partial(torch.tensor, dtype=torch.double, device='hpu') + + @property + def FloatTensor(self): + return functools.partial(torch.tensor, dtype=torch.float, device='hpu') + + @property + def HalfTensor(self): + return functools.partial(torch.tensor, dtype=torch.half, device='hpu') + + @property + def IntTensor(self): + return functools.partial(torch.tensor, dtype=torch.int, device='hpu') + + @property + def LongTensor(self): + return functools.partial(torch.tensor, dtype=torch.long, device='hpu') + + def pin_memory(self, tensor, align_bytes=1): + return tensor.pin_memory(self.device()) + + def is_pinned(self, tensor): + return tensor.is_pinned() + + def on_accelerator(self, tensor): + device_str = str(tensor.device) + if device_str.startswith('hpu:'): + return True + else: + return False + + def op_builder_dir(self): + try: + # is op_builder from deepspeed or a 3p version? this should only succeed if it's deepspeed + # if successful this also means we're doing a local install and not JIT compile path + from op_builder import __deepspeed__ # noqa: F401 # type: ignore + return "op_builder.hpu" + except ImportError: + return "deepspeed.ops.op_builder.hpu" + + # dict that holds class name <--> class type mapping i.e. + # 'AsyncIOBuilder': + # this dict will be filled at init stage + class_dict = None + + def _lazy_init_class_dict(self): + if self.class_dict is not None: + return + else: + self.class_dict = {} + # begin initialize for create_op_builder() + # put all valid class name <--> class type mapping into class_dict + op_builder_dir = self.op_builder_dir() + op_builder_module = importlib.import_module(op_builder_dir) + op_builder_absolute_path = os.path.dirname(op_builder_module.__file__) + for _, module_name, _ in pkgutil.iter_modules([op_builder_absolute_path]): + # avoid self references, + # skip sub_directories which contains ops for other backend(cpu, npu, etc.). + if module_name != 'all_ops' and module_name != 'builder' and not os.path.isdir( + os.path.join(op_builder_absolute_path, module_name)): + module = importlib.import_module("{}.{}".format(op_builder_dir, module_name)) + for member_name in module.__dir__(): + if member_name.endswith( + 'Builder' + ) and member_name != "OpBuilder" and member_name != "CPUOpBuilder" and member_name != "TorchCPUOpBuilder": # avoid abstract classes + if not member_name in self.class_dict: + self.class_dict[member_name] = getattr(module, member_name) + # end initialize for create_op_builder() + + # create an instance of op builder and return, name specified by class_name + def create_op_builder(self, class_name): + self._lazy_init_class_dict() + if class_name in self.class_dict: + return self.class_dict[class_name]() + else: + return None + + # return an op builder class, name specified by class_name + def get_op_builder(self, class_name): + self._lazy_init_class_dict() + if class_name in self.class_dict: + return self.class_dict[class_name] + else: + return self.class_dict['NotImplementedBuilder'] if 'NotImplementedBuilder' in self.class_dict else None + + def build_extension(self): + from torch.utils.cpp_extension import BuildExtension + return BuildExtension + + def export_envs(self): + return [] + + def visible_devices_envs(self): + # Current way deepspeed set this env var is not applicable with all HPU instances + # User has to follow instructions in: + # https://docs.habana.ai/en/latest/PyTorch/Reference/PT_Multiple_Tenants_on_HPU/Multiple_Workloads_Single_Docker.html + # keeping CUDA_VISIBLE_DEVICES + return ['CUDA_VISIBLE_DEVICES'] #['HABANA_VISIBLE_MODULES'] + + def set_visible_devices_envs(self, current_env, local_accelerator_ids): + for env in self.visible_devices_envs(): + current_env[env] = ",".join(map(str, local_accelerator_ids)) + + def get_compile_backend(self): + return self._compile_backend + + def set_compile_backend(self, backend): + supported_backends = torch._dynamo.list_backends(exclude_tags=()) + if backend in supported_backends: + self._compile_backend = backend + else: + raise ValueError( + f"{backend} not supported by {self.device_name()}. Supported Backends are {supported_backends}") diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/mlu_accelerator.py b/lib/python3.12/site-packages/deepspeed/accelerator/mlu_accelerator.py new file mode 100644 index 0000000000000000000000000000000000000000..bef716f0ee4e487f7f3de2f9b944b3db8c837c02 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/accelerator/mlu_accelerator.py @@ -0,0 +1,300 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team +import importlib +import inspect +import functools + +from .abstract_accelerator import DeepSpeedAccelerator +import torch +# During setup stage torch may not be installed, pass on no torch will +# allow op builder related API to be executed. + + +class MLU_Accelerator(DeepSpeedAccelerator): + + def __init__(self): + self._name = 'mlu' + self._communication_backend_name = 'cncl' + self._compile_backend = "inductor" + self.class_dict = None + + def is_synchronized_device(self): + return False + + def use_host_timers(self): + return self.is_synchronized_device() + + def resolves_data_dependency(self): + return self.is_synchronized_device() + + def handles_memory_backpressure(self): + return self.is_synchronized_device() + + # Device APIs + def device_name(self, device_index=None): + if device_index == None: + return 'mlu' + return 'mlu:{}'.format(device_index) + + def device(self, device_index=None): + return torch.mlu.device(device_index) + + def set_device(self, device_index): + torch.mlu.set_device(device_index) + + def current_device(self): + return torch.mlu.current_device() + + def current_device_name(self): + return 'mlu:{}'.format(torch.mlu.current_device()) + + def device_count(self): + return torch.mlu.device_count() + + def synchronize(self, device_index=None): + return torch.mlu.synchronize(device_index) + + # RNG APIs + def random(self): + return torch.random + + def set_rng_state(self, new_state, device_index=None): + if device_index is None: + return torch.mlu.set_rng_state(new_state) + + return torch.mlu.set_rng_state(new_state, device_index) + + def get_rng_state(self, device_index=None): + if device_index is None: + return torch.mlu.get_rng_state() + + return torch.mlu.get_rng_state(device_index) + + def manual_seed(self, seed): + return torch.mlu.manual_seed(seed) + + def manual_seed_all(self, seed): + return torch.mlu.manual_seed_all(seed) + + def initial_seed(self, seed): + return torch.mlu.initial_seed(seed) + + def default_generator(self, device_index): + return torch.mlu.default_generators[device_index] + + # Streams/Events + @property + def Stream(self): + return torch.mlu.Stream + + def stream(self, stream): + return torch.mlu.stream(stream) + + def current_stream(self, device_index=None): + return torch.mlu.current_stream(device_index) + + def default_stream(self, device_index=None): + return torch.mlu.default_stream(device_index) + + @property + def Event(self): + return torch.mlu.Event + + # Memory management + def empty_cache(self): + return torch.mlu.empty_cache() + + def memory_allocated(self, device_index=None): + return torch.mlu.memory_allocated(device_index) + + def max_memory_allocated(self, device_index=None): + return torch.mlu.max_memory_allocated(device_index) + + def reset_max_memory_allocated(self, device_index=None): + return torch.mlu.reset_max_memory_allocated(device_index) + + def memory_cached(self, device_index=None): + return torch.mlu.memory_cached(device_index) + + def max_memory_cached(self, device_index=None): + return torch.mlu.max_memory_cached(device_index) + + def reset_max_memory_cached(self, device_index=None): + return torch.mlu.reset_max_memory_cached(device_index) + + def memory_stats(self, device_index=None): + if hasattr(torch.mlu, 'memory_stats'): + return torch.mlu.memory_stats(device_index) + + def reset_peak_memory_stats(self, device_index=None): + if hasattr(torch.mlu, 'reset_peak_memory_stats'): + return torch.mlu.reset_peak_memory_stats(device_index) + + def memory_reserved(self, device_index=None): + if hasattr(torch.mlu, 'memory_reserved'): + return torch.mlu.memory_reserved(device_index) + + def max_memory_reserved(self, device_index=None): + if hasattr(torch.mlu, 'max_memory_reserved'): + return torch.mlu.max_memory_reserved(device_index) + + def total_memory(self, device_index=None): + return torch.mlu.get_device_properties(device_index).total_memory + + def available_memory(self, device_index=None): + return self.total_memory(device_index) - self.memory_allocated(device_index) + + # Data types + def is_bf16_supported(self): + return torch.mlu.is_bf16_supported() + + def is_fp16_supported(self): + return True + + def supported_dtypes(self): + supported_dtypes = [torch.float] + if self.is_fp16_supported(): + supported_dtypes.append(torch.half) + if self.is_bf16_supported(): + supported_dtypes.append(torch.bfloat16) + return supported_dtypes + + # Misc + def amp(self): + if hasattr(torch.mlu, 'amp'): + return torch.mlu.amp + return None + + def is_available(self): + return torch.mlu.is_available() + + def range_push(self, msg): + if hasattr(torch.mlu.cnpx, 'range_push'): + return torch.mlu.cnpx.range_push(msg) + + def range_pop(self): + if hasattr(torch.mlu.cnpx, 'range_pop'): + return torch.mlu.cnpx.range_pop() + + def lazy_call(self, callback): + return torch.mlu._lazy_call(callback) + + def communication_backend_name(self): + return self._communication_backend_name + + def is_triton_supported(self): + return True + + # Graph operations + def create_graph(self): + torch.mlu.MLUGraph() + + def capture_to_graph(self, graph, pool=None, stream=None): + return torch.mlu.graph(graph, pool, stream) + + def replay_graph(self, graph): + graph.replay() + return + + # Tensor operations + + @property + def BFloat16Tensor(self): + return functools.partial(torch.tensor, dtype=torch.bfloat16, device='mlu') + + @property + def ByteTensor(self): + return functools.partial(torch.tensor, dtype=torch.uint8, device='mlu') + + @property + def DoubleTensor(self): + return functools.partial(torch.tensor, dtype=torch.double, device='mlu') + + @property + def FloatTensor(self): + return functools.partial(torch.tensor, dtype=torch.float, device='mlu') + + @property + def HalfTensor(self): + return functools.partial(torch.tensor, dtype=torch.half, device='mlu') + + @property + def IntTensor(self): + return functools.partial(torch.tensor, dtype=torch.int, device='mlu') + + @property + def LongTensor(self): + return functools.partial(torch.tensor, dtype=torch.long, device='mlu') + + def pin_memory(self, tensor): + return tensor.pin_memory() + + def is_pinned(self, tensor): + return tensor.is_pinned() + + def on_accelerator(self, tensor): + device_str = str(tensor.device) + if device_str.startswith('mlu:'): + return True + else: + return False + + def op_builder_dir(self): + try: + # is op_builder from deepspeed or a 3p version? this should only succeed if it's deepspeed + # if successful this also means we're doing a local install and not JIT compile path + from op_builder import __deepspeed__ # noqa: F401 # type: ignore + return "op_builder.mlu" + except ImportError: + return "deepspeed.ops.op_builder.mlu" + + def _lazy_init_class_dict(self): + if self.class_dict: + return + + op_builder_module = importlib.import_module(self.op_builder_dir()) + + # get op builder class from op_builder/mlu/__init__.py + self.class_dict = {} + for class_name, class_obj in inspect.getmembers(op_builder_module, inspect.isclass): + self.class_dict[class_name] = class_obj + + # create an instance of op builder and return, name specified by class_name + def create_op_builder(self, class_name): + builder_class = self.get_op_builder(class_name) + return builder_class() + + # return an op builder class, name specified by class_name + def get_op_builder(self, class_name): + self._lazy_init_class_dict() + if class_name in self.class_dict: + return self.class_dict[class_name] + else: + return self.class_dict['NotImplementedBuilder'] + + def build_extension(self): + from torch.utils.cpp_extension import BuildExtension + return BuildExtension + + def export_envs(self): + return ['NEUWARE_HOME', 'CNCL', 'LD_LIBRARY', 'PATH'] + + def visible_devices_envs(self): + return ['MLU_VISIBLE_DEVICES'] + + def set_visible_devices_envs(self, current_env, local_accelerator_ids): + for env in self.visible_devices_envs(): + current_env[env] = ",".join(map(str, local_accelerator_ids)) + + def get_compile_backend(self): + return self._compile_backend + + def set_compile_backend(self, backend): + supported_backends = torch._dynamo.list_backends(exclude_tags=()) + if backend in supported_backends: + self._compile_backend = backend + else: + raise ValueError( + f"{backend} not supported by {self.device_name()}. Supported Backends are {supported_backends }") diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/mps_accelerator.py b/lib/python3.12/site-packages/deepspeed/accelerator/mps_accelerator.py new file mode 100644 index 0000000000000000000000000000000000000000..1083ba14d27ee72bacf72d7144438fe13ae4d29a --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/accelerator/mps_accelerator.py @@ -0,0 +1,281 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from .abstract_accelerator import DeepSpeedAccelerator + +# During setup stage torch may not be installed, pass on no torch will +# allow op builder related API to be executed. +try: + import torch.mps +except ImportError: + pass + + +class MPS_Accelerator(DeepSpeedAccelerator): + + def __init__(self): + self._name = "mps" + self._communication_backend_name = None + self._compile_backend = "inductor" + + def is_synchronized_device(self): + return False + + def use_host_timers(self): + return self.is_synchronized_device() + + def resolves_data_dependency(self): + return self.is_synchronized_device() + + def handles_memory_backpressure(self): + return self.is_synchronized_device() + + # Device APIs + def device_name(self, device_index=None): + if device_index is None: + return "mps" + return "mps:{}".format(device_index) + + def device(self, device_index): + return torch.device("mps", index=0) + + def set_device(self, device_index): + return + + def current_device(self): + return torch.device("mps", index=0) + + def current_device_name(self): + return "mps:0" + + def device_count(self): + return 1 + + def synchronize(self, device_index=None): + return torch.mps.synchronize() + + # RNG APIs + def random(self): + return torch.random + + def set_rng_state(self, new_state, device_index=None): + return torch.mps.set_rng_state(new_state) + + def get_rng_state(self, device_index=None): + return torch.mps.get_rng_state() + + def manual_seed(self, seed): + return torch.mps.manual_seed(seed) + + def manual_seed_all(self, seed): + return torch.mps.manual_seed(seed) + + def seed(self): + return torch.mps.seed() + + def initial_seed(self): + return + + def default_generator(self, device_index): + return + + # Streams/Events + @property + def Stream(self): + return None + + def stream(self, stream): + return None + + def current_stream(self, device_index=None): + return None + + def default_stream(self, device_index=None): + return None + + @property + def Event(self): + return None + + # Memory management + def empty_cache(self): + return torch.mps.empty_cache() + + def memory_allocated(self, device_index=None): + return torch.mps.current_allocated_memory() + + def max_memory_allocated(self, device_index=None): + return torch.mps.driver_allocated_memory() + + def set_per_process_memory_fraction(self, fraction): + return torch.mps.set_per_process_memory_fraction(fraction) + + def reset_max_memory_allocated(self, device_index=None): + return + + def memory_cached(self, device_index=None): + return + + def max_memory_cached(self, device_index=None): + return + + def reset_max_memory_cached(self, device_index=None): + return + + def memory_stats(self, device_index=None): + return + + def reset_peak_memory_stats(self, device_index=None): + return + + def memory_reserved(self, device_index=None): + return + + def max_memory_reserved(self, device_index=None): + return + + def total_memory(self, device_index=None): + return + + def available_memory(self, device_index=None): + return + + # Data types + def is_bf16_supported(self): + return False + + def is_fp16_supported(self): + return False + + def supported_dtypes(self): + return [torch.float] + + # Misc + def amp(self): + return + + def is_available(self): + return hasattr(torch.backends, "mps") and torch.backends.mps.is_available() + + def range_push(self, msg): + return + + def range_pop(self): + return + + def lazy_call(self, callback): + return + + def communication_backend_name(self): + return self._communication_backend_name + + def is_triton_supported(self): + return False + + # Graph operations + def create_graph(self): + return None + + def capture_to_graph(self, graph, pool=None, stream=None): + from deepspeed.runtime.utils import noop_context + return noop_context() + + def replay_graph(self, graph): + return + + # Tensor operations + @property + def BFloat16Tensor(self): + return + + @property + def ByteTensor(self): + return + + @property + def DoubleTensor(self): + return + + @property + def FloatTensor(self): + return + + @property + def HalfTensor(self): + return + + @property + def IntTensor(self): + return + + @property + def LongTensor(self): + return + + def pin_memory(self, tensor, align_bytes=1): + return tensor.pin_memory() + + def is_pinned(self, tensor): + return tensor.is_pinned() + + def on_accelerator(self, tensor): + device_str = str(tensor.device) + if device_str.startswith("mps"): + return True + else: + return False + + def op_builder_dir(self): + try: + # is op_builder from deepspeed or a 3p version? this should only succeed if it's deepspeed + # if successful this also means we're doing a local install and not JIT compile path + from op_builder import __deepspeed__ # noqa: F401 # type: ignore + + return "op_builder" + except ImportError: + return "deepspeed.ops.op_builder" + + # create an instance of op builder, specified by class_name + def create_op_builder(self, op_name): + builder_class = self.get_op_builder(op_name) + if builder_class is not None: + return builder_class() + return None + + # return an op builder class, specified by class_name + def get_op_builder(self, class_name): + from deepspeed.ops.op_builder.cpu import NotImplementedBuilder + + return NotImplementedBuilder + + def build_extension(self): + from torch.utils.cpp_extension import BuildExtension + + return BuildExtension + + def export_envs(self): + return [] + + # TODO: mpu's visible envs is confirmed, keep as CUDA_VISIBLE_DEVICES + def visible_devices_envs(self): + # TODO: could not find visible devices env for mps + return ['CUDA_VISIBLE_DEVICES'] + + def set_visible_devices_envs(self, current_env, local_accelerator_ids): + for env in self.visible_devices_envs(): + current_env[env] = ",".join(map(str, local_accelerator_ids)) + + def get_compile_backend(self): + return self._compile_backend + + def set_compile_backend(self, backend): + supported_backends = torch._dynamo.list_backends(exclude_tags=()) + if backend in supported_backends: + self._compile_backend = backend + else: + raise ValueError( + f"{backend} not supported by {self.device_name()}. Supported Backends are {supported_backends}") diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/npu_accelerator.py b/lib/python3.12/site-packages/deepspeed/accelerator/npu_accelerator.py new file mode 100644 index 0000000000000000000000000000000000000000..9252990d0a94fc121aa2cdefb09a66835755e47e --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/accelerator/npu_accelerator.py @@ -0,0 +1,299 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team +import importlib +import inspect + +from .abstract_accelerator import DeepSpeedAccelerator +# During setup stage torch may not be installed, pass on no torch will +# allow op builder related API to be executed. +try: + import torch.npu +except ImportError: + pass + + +class NPU_Accelerator(DeepSpeedAccelerator): + + def __init__(self): + super().__init__() + self._name = 'npu' + self._communication_backend_name = 'hccl' + self._compile_backend = "inductor" + # dict that holds class name <--> class type mapping i.e. + # 'AsyncIOBuilder': + # this dict will be filled at init stage + self.class_dict = None + + def is_synchronized_device(self): + return False + + def use_host_timers(self): + return self.is_synchronized_device() + + def resolves_data_dependency(self): + return self.is_synchronized_device() + + def handles_memory_backpressure(self): + return self.is_synchronized_device() + + # Device APIs + def device_name(self, device_index=None): + if device_index is None: + return 'npu' + return 'npu:{}'.format(device_index) + + def device(self, device_index=None): + return torch.npu.device(device_index) + + def set_device(self, device_index): + torch.npu.set_device(device_index) + + def current_device(self): + return torch.npu.current_device() + + def current_device_name(self): + return 'npu:{}'.format(torch.npu.current_device()) + + def device_count(self): + return torch.npu.device_count() + + def synchronize(self, device_index=None): + return torch.npu.synchronize(device_index) + + # RNG APIs + def random(self): + return torch.random + + def set_rng_state(self, new_state, device_index=None): + if device_index is None: + return torch.npu.set_rng_state(new_state) + + return torch.npu.set_rng_state(new_state, device_index) + + def get_rng_state(self, device_index=None): + if device_index is None: + return torch.npu.get_rng_state() + + return torch.npu.get_rng_state(device_index) + + def manual_seed(self, seed): + return torch.npu.manual_seed(seed) + + def manual_seed_all(self, seed): + return torch.npu.manual_seed_all(seed) + + def initial_seed(self): + return torch.npu.initial_seed() + + def default_generator(self, device_index): + return torch.npu.default_generators[device_index] + + # Streams/Events + @property + def Stream(self): + return torch.npu.Stream + + def stream(self, stream): + return torch.npu.stream(stream) + + def current_stream(self, device_index=None): + return torch.npu.current_stream(device_index) + + def default_stream(self, device_index=None): + return torch.npu.default_stream(device_index) + + @property + def Event(self): + return torch.npu.Event + + # Memory management + def empty_cache(self): + return torch.npu.empty_cache() + + def memory_allocated(self, device_index=None): + return torch.npu.memory_allocated(device_index) + + def max_memory_allocated(self, device_index=None): + return torch.npu.max_memory_allocated(device_index) + + def reset_max_memory_allocated(self, device_index=None): + return torch.npu.reset_max_memory_allocated(device_index) + + def memory_cached(self, device_index=None): + return torch.npu.memory_cached(device_index) + + def max_memory_cached(self, device_index=None): + return torch.npu.max_memory_cached(device_index) + + def reset_max_memory_cached(self, device_index=None): + return torch.npu.reset_max_memory_cached(device_index) + + def memory_stats(self, device_index=None): + if hasattr(torch.npu, 'memory_stats'): + return torch.npu.memory_stats(device_index) + + def reset_peak_memory_stats(self, device_index=None): + if hasattr(torch.npu, 'reset_peak_memory_stats'): + return torch.npu.reset_peak_memory_stats(device_index) + + def memory_reserved(self, device_index=None): + if hasattr(torch.npu, 'memory_reserved'): + return torch.npu.memory_reserved(device_index) + + def max_memory_reserved(self, device_index=None): + if hasattr(torch.npu, 'max_memory_reserved'): + return torch.npu.max_memory_reserved(device_index) + + def total_memory(self, device_index=None): + return torch.npu.get_device_properties(device_index).total_memory + + def available_memory(self, device_index=None): + return self.total_memory(device_index) - self.memory_allocated(device_index) + + # Data types + def is_bf16_supported(self): + return torch.npu.is_bf16_supported() + + def is_fp16_supported(self): + return True + + def supported_dtypes(self): + return [torch.float, torch.half, torch.bfloat16] + + # Misc + def amp(self): + if hasattr(torch.npu, 'amp'): + return torch.npu.amp + return None + + def is_available(self): + return torch.npu.is_available() + + def range_push(self, msg): + return + + def range_pop(self): + return + + def lazy_call(self, callback): + return torch.npu._lazy_call(callback) + + def communication_backend_name(self): + return self._communication_backend_name + + def is_triton_supported(self): + return False + + # Graph operations + def create_graph(self): + return None + + def capture_to_graph(self, graph, pool=None, stream=None): + from deepspeed.runtime.utils import noop_context + return noop_context() + + def replay_graph(self, graph): + return + + # Tensor operations + + @property + def BFloat16Tensor(self): + return torch.npu.BFloat16Tensor + + @property + def ByteTensor(self): + return torch.npu.ByteTensor + + @property + def DoubleTensor(self): + return torch.npu.DoubleTensor + + @property + def FloatTensor(self): + return torch.npu.FloatTensor + + @property + def HalfTensor(self): + return torch.npu.HalfTensor + + @property + def IntTensor(self): + return torch.npu.IntTensor + + @property + def LongTensor(self): + return torch.npu.LongTensor + + def pin_memory(self, tensor, align_bytes=1): + return tensor.pin_memory() + + def is_pinned(self, tensor): + return tensor.is_pinned() + + def on_accelerator(self, tensor): + device_str = str(tensor.device) + if device_str.startswith('npu:'): + return True + else: + return False + + def op_builder_dir(self): + try: + # is op_builder from deepspeed or a 3p version? this should only succeed if it's deepspeed + # if successful this also means we're doing a local install and not JIT compile path + from op_builder import __deepspeed__ # noqa: F401 # type: ignore + return "op_builder.npu" + except ImportError: + return "deepspeed.ops.op_builder.npu" + + def _lazy_init_class_dict(self): + if self.class_dict: + return + + op_builder_module = importlib.import_module(self.op_builder_dir()) + + # get op builder class from op_builder/npu/__init__.py + self.class_dict = {} + for class_name, class_obj in inspect.getmembers(op_builder_module, inspect.isclass): + self.class_dict[class_name] = class_obj + + # create an instance of op builder and return, name specified by class_name + def create_op_builder(self, class_name): + builder_class = self.get_op_builder(class_name) + return None if builder_class is None else builder_class() + + # return an op builder class, name specified by class_name + def get_op_builder(self, class_name): + self._lazy_init_class_dict() + if class_name in self.class_dict: + return self.class_dict[class_name] + else: + return self.class_dict['NotImplementedBuilder'] if 'NotImplementedBuilder' in self.class_dict else None + + def build_extension(self): + from torch.utils.cpp_extension import BuildExtension + return BuildExtension + + def export_envs(self): + return ['ASCEND', 'HCCL', 'LD_LIBRARY', 'PATH'] + + def visible_devices_envs(self): + return ['ASCEND_RT_VISIBLE_DEVICES'] + + def set_visible_devices_envs(self, current_env, local_accelerator_ids): + for env in self.visible_devices_envs(): + current_env[env] = ",".join(map(str, local_accelerator_ids)) + + def get_compile_backend(self): + return self._compile_backend + + def set_compile_backend(self, backend): + supported_backends = torch._dynamo.list_backends(exclude_tags=()) + if backend in supported_backends: + self._compile_backend = backend + else: + raise ValueError( + f"{backend} not supported by {self.device_name()}. Supported Backends are {supported_backends }") diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/real_accelerator.py b/lib/python3.12/site-packages/deepspeed/accelerator/real_accelerator.py new file mode 100644 index 0000000000000000000000000000000000000000..b04790b1590a05641f5e7d4e171c5d068e63e45f --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/accelerator/real_accelerator.py @@ -0,0 +1,308 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team +import os + +try: + # Importing logger currently requires that torch is installed, hence the try...except + # TODO: Remove logger dependency on torch. + from deepspeed.utils import logger as accel_logger +except ImportError as e: + accel_logger = None + +try: + from accelerator.abstract_accelerator import DeepSpeedAccelerator as dsa1 +except ImportError as e: + dsa1 = None +try: + from deepspeed.accelerator.abstract_accelerator import DeepSpeedAccelerator as dsa2 +except ImportError as e: + dsa2 = None + +SUPPORTED_ACCELERATOR_LIST = ['cuda', 'cpu', 'xpu', 'xpu.external', 'npu', 'mps', 'hpu', 'mlu', 'sdaa'] + +ds_accelerator = None + + +def _validate_accelerator(accel_obj): + # because abstract_accelerator has different path during + # build time (accelerator.abstract_accelerator) + # and run time (deepspeed.accelerator.abstract_accelerator) + # and extension would import the + # run time abstract_accelerator/DeepSpeedAccelerator as its base + # class, so we need to compare accel_obj with both base class. + # if accel_obj is instance of DeepSpeedAccelerator in one of + # accelerator.abstractor_accelerator + # or deepspeed.accelerator.abstract_accelerator, consider accel_obj + # is a conforming object + if not ((dsa1 is not None and isinstance(accel_obj, dsa1)) or (dsa2 is not None and isinstance(accel_obj, dsa2))): + raise AssertionError(f"{accel_obj.__class__.__name__} accelerator is not subclass of DeepSpeedAccelerator") + + # TODO: turn off is_available test since this breaks tests + # assert accel_obj.is_available(), \ + # f'{accel_obj.__class__.__name__} accelerator fails is_available() test' + + +def is_current_accelerator_supported(): + return get_accelerator().device_name() in SUPPORTED_ACCELERATOR_LIST + + +def get_accelerator(): + global ds_accelerator + if ds_accelerator is not None: + return ds_accelerator + + accelerator_name = None + ds_set_method = None + # 1. Detect whether there is override of DeepSpeed accelerators from environment variable. + if "DS_ACCELERATOR" in os.environ.keys(): + accelerator_name = os.environ["DS_ACCELERATOR"] + if accelerator_name == "xpu": + try: + import intel_extension_for_pytorch as ipex + assert ipex._C._has_xpu(), "XPU_Accelerator requires an intel_extension_for_pytorch that supports XPU." + except ImportError as e: + raise ValueError( + f"XPU_Accelerator requires intel_extension_for_pytorch, which is not installed on this system.") + elif accelerator_name == "xpu.external": + try: + import intel_extension_for_deepspeed # noqa: F401 # type: ignore + except ImportError as e: + raise ValueError( + f"XPU_Accelerator external requires intel_extension_for_deepspeed, which is not installed on this system." + ) + elif accelerator_name == "cpu": + pass + elif accelerator_name == "npu": + try: + import torch_npu # noqa: F401 # type: ignore + except ImportError as e: + raise ValueError(f"NPU_Accelerator requires torch_npu, which is not installed on this system.") + pass + elif accelerator_name == "sdaa": + try: + import torch_sdaa # noqa: F401 # type: ignore + except ImportError as e: + raise ValueError(f"SDAA_Accelerator requires torch_sdaa, which is not installed on this system.") + pass + elif accelerator_name == "mps": + try: + import torch.mps + + # should use torch.mps.is_available() if it exists someday but this is used as proxy + torch.mps.current_allocated_memory() + except (RuntimeError, ImportError) as e: + raise ValueError(f"MPS_Accelerator requires torch.mps, which is not installed on this system.") + elif accelerator_name == "hpu": + try: + import habana_frameworks.torch.hpu # noqa: F401 + except ImportError as e: + raise ValueError( + f"HPU_Accelerator requires habana_frameworks.torch.hpu, which is not installed on this system.") + elif accelerator_name == "mlu": + try: + import torch_mlu # noqa: F401 + except ImportError as e: + raise ValueError(f"MLU_Accelerator requires torch_mlu, which is not installed on this system.") + elif accelerator_name not in SUPPORTED_ACCELERATOR_LIST: + raise ValueError(f'DS_ACCELERATOR must be one of {SUPPORTED_ACCELERATOR_LIST}. ' + f'Value "{accelerator_name}" is not supported') + ds_set_method = "override" + + # 2. If no override, detect which accelerator to use automatically + if accelerator_name is None: + # We need a way to choose among different accelerator types. + # Currently we detect which accelerator extension is installed + # in the environment and use it if the installing answer is True. + # An alternative might be detect whether CUDA device is installed on + # the system but this comes with two pitfalls: + # 1. the system may not have torch pre-installed, so + # get_accelerator().is_available() may not work. + # 2. Some scenario like install on login node (without CUDA device) + # and run on compute node (with CUDA device) may cause mismatch + # between installation time and runtime. + + try: + from intel_extension_for_deepspeed import XPU_Accelerator # noqa: F401,F811 # type: ignore + accelerator_name = "xpu.external" + except ImportError as e: + pass + if accelerator_name is None: + try: + import intel_extension_for_pytorch as ipex + + if ipex._C._has_xpu(): + accelerator_name = "xpu" + except ImportError as e: + pass + if accelerator_name is None: + try: + import torch + + # torch.xpu will be supported in upstream pytorch-2.8. + # Currently we can run on xpu device only using pytorch, + # also reserve the old path using ipex when the torch version is old. + if hasattr(torch, 'xpu'): + if torch.cuda.device_count() == 0: #ignore-cuda + if torch.xpu.device_count() > 0 and torch.xpu.is_available(): + accelerator_name = "xpu" + else: + pass + except ImportError as e: + pass + if accelerator_name is None: + try: + import torch_npu # noqa: F401,F811 # type: ignore + + accelerator_name = "npu" + except ImportError as e: + pass + if accelerator_name is None: + try: + import torch_sdaa # noqa: F401,F811 # type: ignore + + accelerator_name = "sdaa" + except ImportError as e: + pass + if accelerator_name is None: + try: + import torch.mps + + # should use torch.mps.is_available() if it exists someday but this is used as proxy + torch.mps.current_allocated_memory() + accelerator_name = "mps" + except (RuntimeError, ImportError) as e: + pass + if accelerator_name is None: + try: + import habana_frameworks.torch.hpu # noqa: F401,F811 + + accelerator_name = "hpu" + except ImportError as e: + pass + if accelerator_name is None: + try: + import torch_mlu # noqa: F401,F811 + + accelerator_name = "mlu" + except ImportError as e: + pass + if accelerator_name is None: + try: + import torch + + # Determine if we are on a GPU or x86 CPU with torch. + # "torch.cuda.is_available()" provides a stronger guarantee, #ignore-cuda + # ensuring that we are free from CUDA initialization errors. + # While "torch.cuda.device_count() > 0" check ensures that #ignore-cuda + # we won't try to do any CUDA calls when no device is available + # For reference: https://github.com/deepspeedai/DeepSpeed/pull/6810 + if torch.cuda.device_count() > 0 and torch.cuda.is_available(): #ignore-cuda + accelerator_name = "cuda" + except (RuntimeError, ImportError) as e: + # TODO need a more decent way to detect which accelerator to use, consider using nvidia-smi command for detection + pass + if accelerator_name is None: + # borrow this log from PR#5084 + if accel_logger is not None: + accel_logger.warning( + "Setting accelerator to CPU. If you have GPU or other accelerator, we were unable to detect it.") + # cpu added as catch-all when accelerator detection fails + accelerator_name = "cpu" + + ds_set_method = "auto detect" + + # 3. Set ds_accelerator accordingly + if accelerator_name == "cuda": + from .cuda_accelerator import CUDA_Accelerator + + ds_accelerator = CUDA_Accelerator() + elif accelerator_name == "cpu": + from .cpu_accelerator import CPU_Accelerator + + ds_accelerator = CPU_Accelerator() + elif accelerator_name == "xpu.external": + # XPU_Accelerator is already imported in detection stage + ds_accelerator = XPU_Accelerator() + elif accelerator_name == "xpu": + from .xpu_accelerator import XPU_Accelerator + + ds_accelerator = XPU_Accelerator() + elif accelerator_name == "npu": + from .npu_accelerator import NPU_Accelerator + + ds_accelerator = NPU_Accelerator() + elif accelerator_name == "sdaa": + from .sdaa_accelerator import SDAA_Accelerator + + ds_accelerator = SDAA_Accelerator() + elif accelerator_name == "mps": + from .mps_accelerator import MPS_Accelerator + + ds_accelerator = MPS_Accelerator() + elif accelerator_name == 'hpu': + from .hpu_accelerator import HPU_Accelerator + + ds_accelerator = HPU_Accelerator() + elif accelerator_name == 'mlu': + from .mlu_accelerator import MLU_Accelerator + + ds_accelerator = MLU_Accelerator() + _validate_accelerator(ds_accelerator) + if accel_logger is not None: + accel_logger.info(f"Setting ds_accelerator to {ds_accelerator._name} ({ds_set_method})") + return ds_accelerator + + +def set_accelerator(accel_obj): + global ds_accelerator + _validate_accelerator(accel_obj) + if accel_logger is not None: + accel_logger.info(f"Setting ds_accelerator to {accel_obj._name} (model specified)") + ds_accelerator = accel_obj + + +""" +-----------[code] test_get.py ----------- +from deepspeed.accelerator import get_accelerator +my_accelerator = get_accelerator() +logger.info(f'{my_accelerator._name=}') +logger.info(f'{my_accelerator._communication_backend=}') +logger.info(f'{my_accelerator.HalfTensor().device=}') +logger.info(f'{my_accelerator.total_memory()=}') +-----------[code] test_get.py ----------- + +---[output] python test_get.py--------- +my_accelerator.name()='cuda' +my_accelerator.communication_backend='nccl' +my_accelerator.HalfTensor().device=device(type='cuda', index=0) +my_accelerator.total_memory()=34089730048 +---[output] python test_get.py--------- + +************************************************************************** +-----------[code] test_set.py ----------- +from deepspeed.accelerator.cuda_accelerator import CUDA_Accelerator +cu_accel = CUDA_Accelerator() +logger.info(f'{id(cu_accel)=}') +from deepspeed.accelerator import set_accelerator, get_accelerator +set_accelerator(cu_accel) + +my_accelerator = get_accelerator() +logger.info(f'{id(my_accelerator)=}') +logger.info(f'{my_accelerator._name=}') +logger.info(f'{my_accelerator._communication_backend=}') +logger.info(f'{my_accelerator.HalfTensor().device=}') +logger.info(f'{my_accelerator.total_memory()=}') +-----------[code] test_set.py ----------- + + +---[output] python test_set.py--------- +id(cu_accel)=139648165478304 +my_accelerator= +my_accelerator.name='cuda' +my_accelerator.communication_backend='nccl' +my_accelerator.HalfTensor().device=device(type='cuda', index=0) +my_accelerator.total_memory()=34089730048 +---[output] python test_set.py--------- +""" diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/sdaa_accelerator.py b/lib/python3.12/site-packages/deepspeed/accelerator/sdaa_accelerator.py new file mode 100644 index 0000000000000000000000000000000000000000..26113d38dd15099a8767b5800c9ee12ad9a8dc2e --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/accelerator/sdaa_accelerator.py @@ -0,0 +1,328 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# BSD 3- Clause License Copyright (c) 2023, Tecorigin Co., Ltd. All rights +# reserved. +# Redistribution and use in source and binary forms, with or without +# modification, are permitted provided that the following conditions are met: +# Redistributions of source code must retain the above copyright notice, +# this list of conditions and the following disclaimer. +# Redistributions in binary form must reproduce the above copyright notice, +# this list of conditions and the following disclaimer in the documentation +# and/or other materials provided with the distribution. +# Neither the name of the copyright holder nor the names of its contributors +# may be used to endorse or promote products derived from this software +# without specific prior written permission. +# +# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" +# AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE +# IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE +# ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE +# LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR +# CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF +# SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS +# INTERRUPTION) +# HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, +# STRICT LIABILITY,OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY +# WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY +# OF SUCH DAMAGE. + +# DeepSpeed Team + +import importlib +import inspect +import functools + +from .abstract_accelerator import DeepSpeedAccelerator +# During setup stage torch may not be installed, pass on no torch will +# allow op builder related API to be executed. +try: + import torch.sdaa +except ImportError: + pass + + +class SDAA_Accelerator(DeepSpeedAccelerator): + + def __init__(self): + self._name = 'sdaa' + self._communication_backend_name = 'tccl' + self._compile_backend = "inductor" + self.class_dict = None + + def is_synchronized_device(self): + return False + + def use_host_timers(self): + return self.is_synchronized_device() + + def resolves_data_dependency(self): + return self.is_synchronized_device() + + def handles_memory_backpressure(self): + return self.is_synchronized_device() + + # Device APIs + def device_name(self, device_index=None): + if device_index is None: + return 'sdaa' + return 'sdaa:{}'.format(device_index) + + def device(self, device_index=None): + return torch.sdaa.device(device_index) + + def set_device(self, device_index): + torch.sdaa.set_device(device_index) + + def current_device(self): + return torch.sdaa.current_device() + + def current_device_name(self): + return 'sdaa:{}'.format(torch.sdaa.current_device()) + + def device_count(self): + return torch.sdaa.device_count() + + def synchronize(self, device_index=None): + return torch.sdaa.synchronize(device_index) + + # RNG APIs + def random(self): + return torch.random + + def set_rng_state(self, new_state, device_index=None): + if device_index is None: + return torch.sdaa.set_rng_state(new_state) + + return torch.sdaa.set_rng_state(new_state, device_index) + + def get_rng_state(self, device_index=None): + if device_index is None: + return torch.sdaa.get_rng_state() + + return torch.sdaa.get_rng_state(device_index) + + def manual_seed(self, seed): + return torch.sdaa.manual_seed(seed) + + def manual_seed_all(self, seed): + return torch.sdaa.manual_seed_all(seed) + + def initial_seed(self): + return torch.sdaa.initial_seed() + + def default_generator(self, device_index): + return torch.sdaa.default_generators[device_index] + + # Streams/Events + @property + def Stream(self): + return torch.sdaa.Stream + + def stream(self, stream): + return torch.sdaa.stream(stream) + + def current_stream(self, device_index=None): + return torch.sdaa.current_stream(device_index) + + def default_stream(self, device_index=None): + return torch.sdaa.default_stream(device_index) + + @property + def Event(self): + return torch.sdaa.Event + + # Memory management + def empty_cache(self): + return torch.sdaa.empty_cache() + + def memory_allocated(self, device_index=None): + return torch.sdaa.memory_allocated(device_index) + + def max_memory_allocated(self, device_index=None): + return torch.sdaa.max_memory_allocated(device_index) + + def reset_max_memory_allocated(self, device_index=None): + return torch.sdaa.reset_max_memory_allocated(device_index) + + def memory_cached(self, device_index=None): + return torch.sdaa.memory_cached(device_index) + + def max_memory_cached(self, device_index=None): + return torch.sdaa.max_memory_cached(device_index) + + def reset_max_memory_cached(self, device_index=None): + return torch.sdaa.reset_max_memory_cached(device_index) + + def memory_stats(self, device_index=None): + if hasattr(torch.sdaa, 'memory_stats'): + return torch.sdaa.memory_stats(device_index) + + def reset_peak_memory_stats(self, device_index=None): + if hasattr(torch.sdaa, 'reset_peak_memory_stats'): + return torch.sdaa.reset_peak_memory_stats(device_index) + + def memory_reserved(self, device_index=None): + if hasattr(torch.sdaa, 'memory_reserved'): + return torch.sdaa.memory_reserved(device_index) + + def max_memory_reserved(self, device_index=None): + if hasattr(torch.sdaa, 'max_memory_reserved'): + return torch.sdaa.max_memory_reserved(device_index) + + def total_memory(self, device_index=None): + return torch.sdaa.get_device_properties(device_index).total_memory + + def available_memory(self, device_index=None): + return self.total_memory(device_index) - self.memory_allocated(device_index) + + # Data types + def is_bf16_supported(self): + return torch.sdaa.is_bf16_supported() + + def is_fp16_supported(self): + return True + + def supported_dtypes(self): + supported_dtypes = [torch.float] + if self.is_fp16_supported(): + supported_dtypes.append(torch.half) + if self.is_bf16_supported(): + supported_dtypes.append(torch.bfloat16) + return supported_dtypes + + # Misc + def amp(self): + if hasattr(torch.sdaa, 'amp'): + return torch.sdaa.amp + return None + + def is_available(self): + return torch.sdaa.is_available() + + def range_push(self, msg): + return + + def range_pop(self): + return + + def lazy_call(self, callback): + return torch.sdaa._lazy_call(callback) + + def communication_backend_name(self): + return self._communication_backend_name + + def is_triton_supported(self): + return False + + # Graph operations + def create_graph(self): + return None + + def capture_to_graph(self, graph, pool=None, stream=None): + from deepspeed.runtime.utils import noop_context + return noop_context() + + def replay_graph(self, graph): + return + + # Tensor operations + + @property + def BFloat16Tensor(self): + return functools.partial(torch.tensor, dtype=torch.bfloat16, device='sdaa') + + @property + def ByteTensor(self): + return functools.partial(torch.tensor, dtype=torch.uint8, device='sdaa') + + @property + def DoubleTensor(self): + return functools.partial(torch.tensor, dtype=torch.double, device='sdaa') + + @property + def FloatTensor(self): + return functools.partial(torch.tensor, dtype=torch.float, device='sdaa') + + @property + def HalfTensor(self): + return functools.partial(torch.tensor, dtype=torch.half, device='sdaa') + + @property + def IntTensor(self): + return functools.partial(torch.tensor, dtype=torch.int, device='sdaa') + + @property + def LongTensor(self): + return functools.partial(torch.tensor, dtype=torch.long, device='sdaa') + + def pin_memory(self, tensor, align_bytes=1): + return tensor.pin_memory() + + def is_pinned(self, tensor): + return tensor.is_pinned() + + def on_accelerator(self, tensor): + device_str = str(tensor.device) + if device_str.startswith('sdaa:'): + return True + else: + return False + + def op_builder_dir(self): + try: + # is op_builder from deepspeed or a 3p version? this should only succeed if it's deepspeed + # if successful this also means we're doing a local install and not JIT compile path + from op_builder import __deepspeed__ # noqa: F401 # type: ignore + return "op_builder.sdaa" + except ImportError: + return "deepspeed.ops.op_builder.sdaa" + + def _lazy_init_class_dict(self): + if self.class_dict: + return + + op_builder_module = importlib.import_module(self.op_builder_dir()) + + # get op builder class from op_builder/sdaa/__init__.py + self.class_dict = {} + for class_name, class_obj in inspect.getmembers(op_builder_module, inspect.isclass): + self.class_dict[class_name] = class_obj + + # create an instance of op builder and return, name specified by class_name + def create_op_builder(self, class_name): + builder_class = self.get_op_builder(class_name) + return builder_class() + + # return an op builder class, name specified by class_name + def get_op_builder(self, class_name): + self._lazy_init_class_dict() + if class_name in self.class_dict: + return self.class_dict[class_name] + else: + return self.class_dict['NotImplementedBuilder'] + + def build_extension(self): + from torch.utils.cpp_extension import BuildExtension + return BuildExtension + + def export_envs(self): + return ['NCCL', 'LD_LIBRARY', 'PATH'] + + def visible_devices_envs(self): + return ['SDAA_VISIBLE_DEVICES'] + + def set_visible_devices_envs(self, current_env, local_accelerator_ids): + for env in self.visible_devices_envs(): + current_env[env] = ",".join(map(str, local_accelerator_ids)) + + def get_compile_backend(self): + return self._compile_backend + + def set_compile_backend(self, backend): + supported_backends = torch._dynamo.list_backends(exclude_tags=()) + if backend in supported_backends: + self._compile_backend = backend + else: + raise ValueError( + f"{backend} not supported by {self.device_name()}. Supported Backends are {supported_backends}") diff --git a/lib/python3.12/site-packages/deepspeed/accelerator/xpu_accelerator.py b/lib/python3.12/site-packages/deepspeed/accelerator/xpu_accelerator.py new file mode 100644 index 0000000000000000000000000000000000000000..1cc436683e907bbb6aff7431730f6fb399c6a5c7 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/accelerator/xpu_accelerator.py @@ -0,0 +1,337 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch +from deepspeed.accelerator.abstract_accelerator import DeepSpeedAccelerator +import functools +import importlib +import inspect + +try: + import oneccl_bindings_for_pytorch # noqa: F401 # type: ignore + oneccl_imported_p = True +except ImportError as e: + oneccl_imported_p = False + +try: + import intel_extension_for_pytorch as ipex # noqa: F401 # type: ignore + ipex_imported_p = True +except ImportError as e: + ipex_imported_p = False + + +class XPU_Accelerator(DeepSpeedAccelerator): + + def __init__(self): + self._name = 'xpu' + if oneccl_imported_p: + self._communication_backend_name = 'ccl' + else: + # changed to xccl if not using torch-CCL on XPU device + self._communication_backend_name = 'xccl' + self._compile_backend = "inductor" + self.aligned_tensors = [] + self.class_dict = None + + def is_synchronized_device(self): + return False + + def use_host_timers(self): + if not ipex_imported_p: + return self.is_synchronized_device() + else: + # WA XPU event will be consolidated in 2.6 + if ipex.__version__ < '2.6': + return True + else: + return self.is_synchronized_device() + + def resolves_data_dependency(self): + return self.is_synchronized_device() + + def handles_memory_backpressure(self): + return self.is_synchronized_device() + + # Device APIs + def device_name(self, device_index=None): + if device_index == None: + return 'xpu' + return 'xpu:{}'.format(device_index) + + def device(self, device_index=None): + return torch.xpu.device(device_index) + + def set_device(self, device_index): + torch.xpu.set_device(device_index) + + def current_device(self): + return torch.xpu.current_device() + + def current_device_name(self): + return 'xpu:{}'.format(torch.xpu.current_device()) + + def device_count(self): + return torch.xpu.device_count() + + def synchronize(self, device_index=None): + return torch.xpu.synchronize(device_index) + + # RNG APIs + def random(self): + return torch.xpu.random + + def set_rng_state(self, new_state, device_index=None): + if device_index == None: + return torch.xpu.set_rng_state(new_state) + return torch.xpu.set_rng_state(new_state, device_index) + + def get_rng_state(self, device_index=None): + if device_index == None: + return torch.xpu.get_rng_state() + return torch.xpu.get_rng_state(device_index) + + def manual_seed(self, seed): + return torch.xpu.manual_seed(seed) + + def manual_seed_all(self, seed): + return torch.xpu.manual_seed_all(seed) + + def initial_seed(self): + return torch.xpu.initial_seed() + + def default_generator(self, device_index): + return torch.xpu.default_generators[device_index] + + # Streams/Events + @property + def Stream(self): + return torch.xpu.Stream + + def stream(self, stream): + return torch.xpu.stream(stream) + + def current_stream(self, device_index=None): + return torch.xpu.current_stream(device_index) + + def default_stream(self, device_index=None): + # torch.xpu does not support the sync behavior of default stream as cuda + # use current_stream as workaround + # see https://pytorch.org/docs/stable/notes/cuda.html#cuda-streams + return torch.xpu.current_stream(device_index) + + @property + def Event(self): + return torch.xpu.Event + + # Memory management + def empty_cache(self): + return torch.xpu.empty_cache() + + def memory_allocated(self, device_index=None): + return torch.xpu.memory_allocated(device_index) + + def max_memory_allocated(self, device_index=None): + return torch.xpu.max_memory_allocated(device_index) + + def reset_max_memory_allocated(self, device_index=None): + return torch.xpu.reset_max_memory_allocated(device_index) + + def memory_cached(self, device_index=None): + return torch.xpu.memory_reserved(device_index) + + def max_memory_cached(self, device_index=None): + return torch.xpu.max_memory_reserved(device_index) + + def reset_max_memory_cached(self, device_index=None): + return torch.xpu.reset_max_memory_reserved(device_index) + + def memory_stats(self, device_index=None): + return torch.xpu.memory_stats(device_index) + + def reset_peak_memory_stats(self, device_index=None): + return torch.xpu.reset_peak_memory_stats(device_index) + + def memory_reserved(self, device_index=None): + return torch.xpu.memory_reserved(device_index) + + def max_memory_reserved(self, device_index=None): + return torch.xpu.max_memory_reserved(device_index) + + def total_memory(self, device_index=None): + return torch.xpu.get_device_properties(device_index).total_memory + + def available_memory(self, device_index=None): + return self.total_memory(device_index) - self.memory_allocated(device_index) + + # Misc + def amp(self): + return torch.xpu.amp + + def is_available(self): + return torch.xpu.is_available() + + def range_push(self, msg): + # TODO itt is currently not supported yet + # return torch.profiler.itt.range_push(msg) + return + + def range_pop(self): + # TODO itt is currently not supported yet + # return torch.profiler.itt.range_pop() + return + + def lazy_call(self, callback): + if hasattr(torch.xpu, "_lazy_call"): + return torch.xpu._lazy_call(callback) + else: + return torch.xpu.lazy_init._lazy_call(callback) + + def communication_backend_name(self): + return self._communication_backend_name + + def is_triton_supported(self): + return False + + # Graph operations + def create_graph(self): + return None + + def capture_to_graph(self, graph, pool=None, stream=None): + from deepspeed.runtime.utils import noop_context + return noop_context() + + def replay_graph(self, graph): + return + + # Data types + def is_bf16_supported(self): + return True + + def is_fp16_supported(self): + return True + + def supported_dtypes(self): + return [torch.float, torch.half, torch.bfloat16] + + # Tensor operations + + @property + def BFloat16Tensor(self): + return functools.partial(torch.tensor, dtype=torch.bfloat16, device=self._name) + + @property + def ByteTensor(self): + return functools.partial(torch.tensor, dtype=torch.uint8, device=self._name) + + @property + def DoubleTensor(self): + return functools.partial(torch.tensor, dtype=torch.double, device=self._name) + + @property + def FloatTensor(self): + return functools.partial(torch.tensor, dtype=torch.float, device=self._name) + + @property + def HalfTensor(self): + return functools.partial(torch.tensor, dtype=torch.half, device=self._name) + + @property + def IntTensor(self): + return functools.partial(torch.tensor, dtype=torch.int, device=self._name) + + @property + def LongTensor(self): + return functools.partial(torch.tensor, dtype=torch.long, device=self._name) + + def pin_memory(self, tensor, align_bytes=1): + if align_bytes == 1: + return tensor.pin_memory(device=self.current_device_name()) + elif align_bytes == 0: + from deepspeed.ops.op_builder.xpu import AsyncIOBuilder + self.aio_handle = AsyncIOBuilder().load().aio_handle(128 * 1024, 8, False, False, False) + aligned_t = self.aio_handle.new_cpu_locked_tensor(tensor.numel(), tensor) + aligned_t = aligned_t[:tensor.numel()].copy_(tensor) + self.aligned_tensors.append([aligned_t.data_ptr(), aligned_t[-1].data_ptr()]) + return aligned_t + + def is_pinned(self, tensor): + if tensor.is_pinned(device=self.current_device_name()): + return True + else: + for begin, end in self.aligned_tensors: + if begin <= tensor.data_ptr() and tensor.data_ptr() <= end: + return True + return False + + def op_builder_dir(self): + try: + # is op_builder from deepspeed or a 3p version? this should only succeed if it's deepspeed + # if successful this also means we're doing a local install and not JIT compile path + from op_builder import __deepspeed__ # noqa: F401 # type: ignore + return "op_builder.xpu" + except ImportError: + return "deepspeed.ops.op_builder.xpu" + + def on_accelerator(self, tensor): + device_str = str(tensor.device) + if device_str.startswith('xpu:'): + return True + else: + return False + + def _lazy_init_class_dict(self): + if self.class_dict: + return + + op_builder_module = importlib.import_module(self.op_builder_dir()) + + # get op builder class from op_builder/xpu/__init__.py + self.class_dict = {} + for class_name, class_obj in inspect.getmembers(op_builder_module, inspect.isclass): + self.class_dict[class_name] = class_obj + + # create an instance of op builder and return, name specified by class_name + def create_op_builder(self, class_name): + builder_class = self.get_op_builder(class_name) + return builder_class() + + # return an op builder class, name specified by class_name + def get_op_builder(self, class_name): + self._lazy_init_class_dict() + if class_name in self.class_dict: + return self.class_dict[class_name] + else: + return self.class_dict['NotImplementedBuilder'] + + def build_extension(self): + if ipex_imported_p: + try: + from intel_extension_for_pytorch.xpu.cpp_extension import DpcppBuildExtension + except ImportError: + from intel_extension_for_pytorch.xpu.utils import DpcppBuildExtension + else: + from torch.utils.cpp_extension import DpcppBuildExtension + return DpcppBuildExtension + + def export_envs(self): + return [] + + def visible_devices_envs(self): + return ['ZE_AFFINITY_MASK'] + + def set_visible_devices_envs(self, current_env, local_accelerator_ids): + for env in self.visible_devices_envs(): + current_env[env] = ",".join(map(str, local_accelerator_ids)) + + def get_compile_backend(self): + return self._compile_backend + + def set_compile_backend(self, backend): + supported_backends = torch._dynamo.list_backends(exclude_tags=()) + if backend in supported_backends: + self._compile_backend = backend + else: + raise ValueError( + f"{backend} not supported by {self.device_name()}. Supported Backends are {supported_backends}") diff --git a/lib/python3.12/site-packages/deepspeed/constants.py b/lib/python3.12/site-packages/deepspeed/constants.py new file mode 100644 index 0000000000000000000000000000000000000000..30135f41b7b68920c32bb2a259a8e431196f4d53 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/constants.py @@ -0,0 +1,21 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import os +from datetime import timedelta + +############################################# +# Torch distributed constants +############################################# +TORCH_DISTRIBUTED_DEFAULT_PORT = 29500 + +# Default process group wide timeout, if applicable. +# This only applies to the gloo and nccl backends +# (only if NCCL_BLOCKING_WAIT or NCCL_ASYNC_ERROR_HANDLING is set to 1). +# To make an attempt at backwards compatibility with THD, we use an +# extraordinarily high default timeout, given that THD did not have timeouts. +default_pg_timeout = timedelta(minutes=int(os.getenv("DEEPSPEED_TIMEOUT", default=30))) +INFERENCE_GENERIC_MODE = 'generic' +INFERENCE_SPECIALIZED_MODE = 'specialized' diff --git a/lib/python3.12/site-packages/deepspeed/env_report.py b/lib/python3.12/site-packages/deepspeed/env_report.py new file mode 100644 index 0000000000000000000000000000000000000000..37e33b1e873972e10194ef70a7fd4470af6a7e30 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/env_report.py @@ -0,0 +1,195 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import os +import torch +import deepspeed +import subprocess +import argparse +from .ops.op_builder.all_ops import ALL_OPS +from .git_version_info import installed_ops, torch_info, accelerator_name +from deepspeed.accelerator import get_accelerator + +GREEN = '\033[92m' +RED = '\033[91m' +YELLOW = '\033[93m' +END = '\033[0m' +SUCCESS = f"{GREEN} [SUCCESS] {END}" +OKAY = f"{GREEN}[OKAY]{END}" +WARNING = f"{YELLOW}[WARNING]{END}" +FAIL = f'{RED}[FAIL]{END}' +INFO = '[INFO]' + +color_len = len(GREEN) + len(END) +okay = f"{GREEN}[OKAY]{END}" +warning = f"{YELLOW}[WARNING]{END}" + + +def op_report(verbose=True): + max_dots = 23 + max_dots2 = 11 + h = ["op name", "installed", "compatible"] + print("-" * (max_dots + max_dots2 + len(h[0]) + len(h[1]))) + print("DeepSpeed C++/CUDA extension op report") + print("-" * (max_dots + max_dots2 + len(h[0]) + len(h[1]))) + + print("NOTE: Ops not installed will be just-in-time (JIT) compiled at\n" + " runtime if needed. Op compatibility means that your system\n" + " meet the required dependencies to JIT install the op.") + + print("-" * (max_dots + max_dots2 + len(h[0]) + len(h[1]))) + print("JIT compiled ops requires ninja") + ninja_status = OKAY if ninja_installed() else FAIL + print('ninja', "." * (max_dots - 5), ninja_status) + print("-" * (max_dots + max_dots2 + len(h[0]) + len(h[1]))) + print(h[0], "." * (max_dots - len(h[0])), h[1], "." * (max_dots2 - len(h[1])), h[2]) + print("-" * (max_dots + max_dots2 + len(h[0]) + len(h[1]))) + installed = f"{GREEN}[YES]{END}" + no = f"{YELLOW}[NO]{END}" + for op_name, builder in ALL_OPS.items(): + dots = "." * (max_dots - len(op_name)) + is_compatible = OKAY if builder.is_compatible(verbose) else no + is_installed = installed if installed_ops.get(op_name, + False) and accelerator_name == get_accelerator()._name else no + dots2 = '.' * ((len(h[1]) + (max_dots2 - len(h[1]))) - (len(is_installed) - color_len)) + print(op_name, dots, is_installed, dots2, is_compatible) + print("-" * (max_dots + max_dots2 + len(h[0]) + len(h[1]))) + + +def ninja_installed(): + try: + import ninja # noqa: F401 # type: ignore + except ImportError: + return False + return True + + +def nvcc_version(): + import torch.utils.cpp_extension + cuda_home = torch.utils.cpp_extension.CUDA_HOME + if cuda_home is None: + return f"{RED} [FAIL] cannot find CUDA_HOME via torch.utils.cpp_extension.CUDA_HOME={torch.utils.cpp_extension.CUDA_HOME} {END}" + try: + output = subprocess.check_output([cuda_home + "/bin/nvcc", "-V"], universal_newlines=True) + except FileNotFoundError: + return f"{RED} [FAIL] nvcc missing {END}" + output_split = output.split() + release_idx = output_split.index("release") + release = output_split[release_idx + 1].replace(',', '').split(".") + return ".".join(release) + + +def installed_cann_path(): + if "ASCEND_HOME_PATH" in os.environ or os.path.exists(os.environ["ASCEND_HOME_PATH"]): + return os.environ["ASCEND_HOME_PATH"] + return None + + +def installed_cann_version(): + import re + ascend_path = installed_cann_path() + if ascend_path is None: + return f"CANN_HOME does not exist, unable to compile NPU op(s)" + cann_version = "" + for dirpath, _, filenames in os.walk(os.path.realpath(ascend_path)): + if cann_version: + break + install_files = [file for file in filenames if re.match(r"ascend_.*_install\.info", file)] + if install_files: + filepath = os.path.join(dirpath, install_files[0]) + with open(filepath, "r") as f: + for line in f: + if line.find("version") != -1: + cann_version = line.strip().split("=")[-1] + break + return cann_version + + +def get_shm_size(): + try: + shm_stats = os.statvfs('/dev/shm') + except (OSError, FileNotFoundError, ValueError, AttributeError): + return "UNKNOWN", None + + shm_size = shm_stats.f_frsize * shm_stats.f_blocks + shm_hbytes = human_readable_size(shm_size) + warn = [] + if shm_size < 512 * 1024**2: + warn.append( + f" {YELLOW} [WARNING] /dev/shm size might be too small, if running in docker increase to at least --shm-size='1gb' {END}" + ) + if get_accelerator().communication_backend_name() == "nccl": + warn.append( + f" {YELLOW} [WARNING] see more details about NCCL requirements: https://docs.nvidia.com/deeplearning/nccl/user-guide/docs/troubleshooting.html#sharing-data {END}" + ) + return shm_hbytes, warn + + +def human_readable_size(size): + units = ['B', 'KB', 'MB', 'GB', 'TB'] + i = 0 + while size >= 1024 and i < len(units) - 1: + size /= 1024 + i += 1 + return f'{size:.2f} {units[i]}' + + +def debug_report(): + max_dots = 33 + + report = [("torch install path", torch.__path__), ("torch version", torch.__version__), + ("deepspeed install path", deepspeed.__path__), + ("deepspeed info", f"{deepspeed.__version__}, {deepspeed.__git_hash__}, {deepspeed.__git_branch__}")] + if get_accelerator().device_name() == 'cuda': + hip_version = getattr(torch.version, "hip", None) + report.extend([("torch cuda version", torch.version.cuda), ("torch hip version", hip_version), + ("nvcc version", (None if hip_version else nvcc_version())), + ("deepspeed wheel compiled w.", f"torch {torch_info['version']}, " + + (f"hip {torch_info['hip_version']}" if hip_version else f"cuda {torch_info['cuda_version']}")) + ]) + elif get_accelerator().device_name() == 'npu': + import torch_npu + report.extend([("deepspeed wheel compiled w.", f"torch {torch_info['version']}"), + ("torch_npu install path", torch_npu.__path__), ("torch_npu version", torch_npu.__version__), + ("ascend_cann version", installed_cann_version())]) + else: + report.extend([("deepspeed wheel compiled w.", f"torch {torch_info['version']} ")]) + + report.append(("shared memory (/dev/shm) size", get_shm_size())) + + print("DeepSpeed general environment info:") + for name, value in report: + warns = [] + if isinstance(value, tuple): + value, warns = value + print(name, "." * (max_dots - len(name)), value) + if warns: + for warn in warns: + print(warn) + + +def parse_arguments(): + parser = argparse.ArgumentParser() + parser.add_argument('--hide_operator_status', + action='store_true', + help='Suppress display of installation and compatibility statuses of DeepSpeed operators. ') + parser.add_argument('--hide_errors_and_warnings', action='store_true', help='Suppress warning and error messages.') + args = parser.parse_args() + return args + + +def main(hide_operator_status=False, hide_errors_and_warnings=False): + if not hide_operator_status: + op_report(verbose=not hide_errors_and_warnings) + debug_report() + + +def cli_main(): + args = parse_arguments() + main(hide_operator_status=args.hide_operator_status, hide_errors_and_warnings=args.hide_errors_and_warnings) + + +if __name__ == "__main__": + main() diff --git a/lib/python3.12/site-packages/deepspeed/git_version_info.py b/lib/python3.12/site-packages/deepspeed/git_version_info.py new file mode 100644 index 0000000000000000000000000000000000000000..70c536d2f78eee24c21db985c43fc9a3c017d4ee --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/git_version_info.py @@ -0,0 +1,31 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +try: + # This is populated by setup.py + from .git_version_info_installed import * # noqa: F401 # type: ignore +except ModuleNotFoundError: + import os + if os.path.isfile('version.txt'): + # Will be missing from checkouts that haven't been installed (e.g., readthedocs) + version = open('version.txt', 'r').read().strip() + else: + version = "0.0.0" + git_hash = '[none]' + git_branch = '[none]' + + from .ops.op_builder.all_ops import ALL_OPS + installed_ops = dict.fromkeys(ALL_OPS.keys(), False) + accelerator_name = "" + torch_info = {'version': "0.0", "cuda_version": "0.0", "hip_version": "0.0"} + +# compatible_ops list is recreated for each launch +from .ops.op_builder.all_ops import ALL_OPS + +compatible_ops = dict.fromkeys(ALL_OPS.keys(), False) +for op_name, builder in ALL_OPS.items(): + op_compatible = builder.is_compatible() + compatible_ops[op_name] = op_compatible + compatible_ops["deepspeed_not_implemented"] = False diff --git a/lib/python3.12/site-packages/deepspeed/git_version_info_installed.py b/lib/python3.12/site-packages/deepspeed/git_version_info_installed.py new file mode 100644 index 0000000000000000000000000000000000000000..b0e73829806780846ce9d7115057e6c021de76ce --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/git_version_info_installed.py @@ -0,0 +1,6 @@ +version='0.17.0' +git_hash='unknown' +git_branch='unknown' +installed_ops={'deepspeed_not_implemented': False, 'async_io': False, 'deepspeed_ccl_comm': False, 'deepspeed_shm_comm': False, 'cpu_adam': False, 'fused_adam': False} +accelerator_name='cpu' +torch_info={'version': '0.0', 'bf16_support': False, 'cuda_version': '0.0', 'nccl_version': '0.0', 'hip_version': '0.0'} diff --git a/lib/python3.12/site-packages/deepspeed/inference/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..cdd00fec935b826c821c8ca9d62e2711a91ea811 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/__init__.py @@ -0,0 +1,7 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team +from .v2 import RaggedInferenceEngineConfig, DeepSpeedTPConfig +from .v2.engine_v2 import InferenceEngineV2 +from .v2 import build_hf_engine, build_engine_from_ds_checkpoint diff --git a/lib/python3.12/site-packages/deepspeed/inference/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2103ba315840fffa3413005575be9f2d1c68b90a Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/__pycache__/config.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/__pycache__/config.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f170894f8bb3127ef56f27a12f42ccaa100d4bc7 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/__pycache__/config.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/__pycache__/engine.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/__pycache__/engine.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ea3e3b99342102a0ed7a70d2def4a1ba3a950d41 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/__pycache__/engine.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/config.py b/lib/python3.12/site-packages/deepspeed/inference/config.py new file mode 100644 index 0000000000000000000000000000000000000000..6df61f7c8841df35540230cf26874418f58cf5a3 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/config.py @@ -0,0 +1,323 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch +import deepspeed +from pydantic import Field, field_validator +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from deepspeed.runtime.zero.config import DeepSpeedZeroConfig +from typing import Dict, Union, Optional +from enum import Enum + + +class DtypeEnum(Enum): + fp16 = (torch.float16, "torch.float16", "fp16", "float16", "half") + fp32 = (torch.float32, "torch.float32", "fp32", "float32", "float") + bf16 = (torch.bfloat16, "torch.bfloat16", "bf16", "bfloat16", "bfloat") + int8 = (torch.int8, "torch.int8", "int8") + + @classmethod + def from_str(cls, value: str): + for dtype in cls: + if value in dtype.value: + return dtype + raise ValueError(f"'{value}' is not a valid DtypeEnum") + + +class MoETypeEnum(str, Enum): + residual = "residual" + standard = "standard" + + +class DeepSpeedTPConfig(DeepSpeedConfigModel): + """ Configure tensor parallelism settings """ + + enabled: bool = True + """ Turn tensor parallelism on/off. """ + + tp_size: int = 1 + """ Number of devices to split the model across using tensor parallelism. """ + + tp_grain_size: int = 64 + "Desired MLP/lm_head tp size granularity. DNN library favors tensor size in granularity of power of 2, we pick 64 as a default size." + + mpu: object = None + """ + A model parallelism unit object that implements + ``get_{model,data}_parallel_{rank,group,world_size}()``. + """ + + tp_group: object = None + + +class DeepSpeedMoEConfig(DeepSpeedConfigModel): + """ Sets parameters for MoE """ + + enabled: bool = True + ep_size: int = 1 + """ + The expert-parallelism size which is used for partitioning the experts + across the GPUs in the expert-parallel group. + """ + + moe_experts: list = Field([1], alias="num_experts") + """ The global number of experts used in an MoE layer. """ + + type: MoETypeEnum = MoETypeEnum.standard + """ + Specify the type of MoE layer. We have two types of MoE layer: 'Standard' + and 'Residual'. + """ + + ep_mp_group: object = None + ep_group: object = Field(None, alias="expert_group") + + +class QuantTypeEnum(str, Enum): + asym = "asymmetric" + sym = "symmetric" + + +class BaseQuantConfig(DeepSpeedConfigModel): + enabled: bool = True + num_bits: int = 8 + q_type: QuantTypeEnum = QuantTypeEnum.sym + q_groups: int = 1 + + +class WeightQuantConfig(BaseQuantConfig): + enabled: bool = True + quantized_initialization: Dict = {} + post_init_quant: Dict = {} + + +class ActivationQuantConfig(BaseQuantConfig): + enabled: bool = True + + +class QKVQuantConfig(DeepSpeedConfigModel): + enabled: bool = True + + +class QuantizationConfig(DeepSpeedConfigModel): + enabled: bool = True + activation: ActivationQuantConfig = ActivationQuantConfig() + weight: WeightQuantConfig = WeightQuantConfig() + qkv: QKVQuantConfig = QKVQuantConfig() + + +# todo: brainstorm on how to do ckpt loading for DS inference +class InferenceCheckpointConfig(DeepSpeedConfigModel): + checkpoint_dir: Optional[str] = None + save_mp_checkpoint_path: Optional[str] = None + base_dir: Optional[str] = None + + +class DeepSpeedInferenceConfig(DeepSpeedConfigModel): + """ Sets parameters for DeepSpeed Inference Engine. """ + + replace_with_kernel_inject: bool = Field(False, alias="kernel_inject") + """ + Set to true to inject inference kernels for models such as, Bert, GPT2, + GPT-Neo and GPT-J. Otherwise, the injection_dict provides the names of two + linear layers as a tuple: + `(attention_output projection, transformer output projection)` + """ + + dtype: torch.dtype = torch.float16 + """ + Desired model data type, will convert model to this type. + Supported target types: `torch.half`, `torch.int8`, `torch.float` + """ + + tensor_parallel: DeepSpeedTPConfig = Field({}, alias="tp") + """ + Configuration for tensor parallelism used to split the model across several + GPUs. Expects a dictionary containing values for :any:`DeepSpeedTPConfig`. + """ + + enable_cuda_graph: bool = False + """ + Use this flag for capturing the CUDA-Graph of the inference ops, so that it + can run faster using the graph replay method. + """ + + use_triton: bool = False + """ + Use this flag to use triton kernels for inference ops. + """ + + triton_autotune: bool = False + """ + Use this flag to enable triton autotuning. + Turning it on is better for performance but increase the 1st runtime for + autotuning. + """ + + zero: DeepSpeedZeroConfig = {} + """ + ZeRO configuration to use with the Inference Engine. Expects a dictionary + containing values for :any:`DeepSpeedZeroConfig`. + """ + + triangular_masking: bool = Field(True, alias="tm") + """ + Controls the type of masking for attention scores in transformer layer. + Note that the masking is application specific. + """ + + moe: Union[bool, DeepSpeedMoEConfig] = {} + """ + Specify if the type of Transformer is MoE. Expects a dictionary containing + values for :any:`DeepSpeedMoEConfig`. + """ + + keep_module_on_host: bool = False + """ + When loading checkpoints to model parameters, they are moved to the device. In very large models + this might fill the device and cause OOM. Setting this flag to true, will keep checkpoints on + host and not move them directly to the device (giving an option to quantize checkpoint data before + moving it to the device for example). + Set only for models with injection policies and auto TP. + """ + + quant: QuantizationConfig = {} + """ + NOTE: only works for int8 dtype. + Quantization settings used for quantizing your model using the MoQ. The + setting can be one element or a tuple. If one value is passed in, we + consider it as the number of groups used in quantization. A tuple is passed + in if we want to mention that there is extra-grouping for the MLP part of a + Transformer layer (e.g. (True, 8) shows we quantize the model using 8 + groups for all the network except the MLP part that we use 8 extra + grouping). Expects a dictionary containing values for + :any:`QuantizationConfig`. + """ + + #todo: refactor the following 3 into the new checkpoint_config + checkpoint: Optional[Union[str, Dict]] = None + """ + Path to deepspeed compatible checkpoint or path to JSON with load policy. + """ + + base_dir: str = "" + """ + This shows the root directory under which all the checkpoint files exists. + This can be passed through the json config too. + """ + + set_empty_params: bool = False + """ + specifying whether the inference-module is created with empty or real Tensor + """ + + save_mp_checkpoint_path: Optional[str] = None + """ + The path for which we want to save the loaded model with a checkpoint. This + feature is used for adjusting the parallelism degree to help alleviate the + model loading overhead. It does not save any new checkpoint if no path is + passed. + """ + + checkpoint_config: InferenceCheckpointConfig = Field({}, alias="ckpt_config") + """ + TODO: Add docs. Expects a dictionary containing values for + :any:`InferenceCheckpointConfig`. + """ + + return_tuple: bool = True + """ + Specify whether or not the transformer layers need to return a tuple or a + Tensor. + """ + + training_mp_size: int = 1 + """ + If loading a checkpoint this is the mp size that it was trained with, it + may be different than what the mp size that you want to use during + inference. + """ + + replace_method: str = Field( + "auto", + json_schema_extra={ + "deprecated": True, + "deprecated_msg": "This parameter is no longer needed, please remove from your call to DeepSpeed-inference" + }) + + injection_policy: Optional[Dict] = Field(None, alias="injection_dict") + """ + Dictionary mapping a client nn.Module to its corresponding injection + policy. e.g., `{BertLayer : deepspeed.inference.HFBertLayerPolicy}` + """ + + injection_policy_tuple: Optional[tuple] = None + """ TODO: Add docs """ + + config: Optional[Dict] = Field(None, alias="args") # todo: really no need for this field if we can refactor + + max_out_tokens: int = Field(1024, alias="max_tokens") + """ + This argument shows the maximum number of tokens inference-engine can work + with, including the input and output tokens. Please consider increasing it + to the required token-length required for your use-case. + """ + + min_out_tokens: int = Field(1, alias="min_tokens") + """ + This argument communicates to the runtime the minimum number of tokens you + expect you will need to generate. This will cause the runtime to error + if it unable to provide this and provide context on the memory pressure + rather than seg-faulting or providing corrupted output. + """ + + transposed_mode: bool = Field(False, alias="transposed_mode") + + mp_size: int = Field(1, json_schema_extra={"deprecated": True, "new_param": "tensor_parallel.tp_size"}) + """ + Desired model parallel size, default is 1 meaning no model parallelism. + Deprecated, please use the ``tensor_parallel` config to control model + parallelism. + """ + mpu: object = Field(None, json_schema_extra={"deprecated": True, "new_param": "tensor_parallel.mpu"}) + ep_size: int = Field(1, json_schema_extra={"deprecated": True, "new_param": "moe.ep_size"}) + ep_group: object = Field(None, + alias="expert_group", + json_schema_extra={ + "deprecated": True, + "new_param": "moe.ep_group" + }) + ep_mp_group: object = Field(None, + alias="expert_mp_group", + json_schema_extra={ + "deprecated": True, + "new_param": "moe.ep_mp_group" + }) + moe_experts: list = Field([1], json_schema_extra={"deprecated": True, "new_param": "moe.moe_experts"}) + moe_type: MoETypeEnum = Field(MoETypeEnum.standard, + json_schema_extra={ + "deprecated": True, + "new_param": "moe.type" + }) + + @field_validator("dtype", mode="before") + def validate_dtype(cls, field_value, values): + if isinstance(field_value, str): + return DtypeEnum.from_str(field_value).value[0] + if isinstance(field_value, torch.dtype): + return field_value + raise TypeError(f"Invalid type for dtype: {type(field_value)}") + + @field_validator("moe") + def moe_backward_compat(cls, field_value, values): + if isinstance(field_value, bool): + return DeepSpeedMoEConfig(moe=field_value) + return field_value + + @field_validator("use_triton") + def has_triton(cls, field_value, values): + if field_value and not deepspeed.HAS_TRITON: + raise ValueError('Triton needs to be installed to use deepspeed with triton kernels') + return field_value diff --git a/lib/python3.12/site-packages/deepspeed/inference/engine.py b/lib/python3.12/site-packages/deepspeed/inference/engine.py new file mode 100644 index 0000000000000000000000000000000000000000..0a74d19e91f5a6bfe8206db0503ca3efd95b835a --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/engine.py @@ -0,0 +1,625 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch +import time +import os +import deepspeed +from deepspeed import comm as dist +from deepspeed.utils.logging import log_dist + +from torch.nn.modules import Module +from packaging import version as pkg_version +from deepspeed.runtime.checkpoint_engine.torch_checkpoint_engine import TorchCheckpointEngine +from deepspeed.utils.timer import SynchronizedWallClockTimer +from deepspeed.runtime.compiler import is_compile_supported +from ..runtime.state_dict_factory import SDLoaderFactory +from ..runtime.weight_quantizer import WeightQuantization +from ..module_inject import replace_transformer_layer, generic_injection +from ..comm.comm import init_distributed +from ..pipe import PipelineModule +from ..moe.utils import has_moe_layers +from ..module_inject import LinearAllreduce, LinearLayer, Normalize, ReplaceWithTensorSlicing +from deepspeed.accelerator import get_accelerator +from ..module_inject.policy import TransformerPolicy +from ..module_inject.auto_tp import AutoTP + +from ..module_inject.replace_policy import generic_policies +from ..module_inject.auto_tp_model_utils import build_bloom_alibi_tensor, build_mpt_atten_bias_tensor, build_mpt_alibi_tensor, get_alibi_mask +from ..ops.transformer.inference.ds_attention import DeepSpeedSelfAttention +from ..model_implementations.transformers.ds_transformer import DeepSpeedTransformerInference + +DS_INFERENCE_ENABLED = False +from torch import nn + +INFERENCE_MODEL_TIMER = "model-forward-inference" + + +class InferenceEngine(Module): + inference_mp_group = None + inference_ep_group = None + expert_mp_group = None + + def __init__(self, model, config): + """ + Args: + model: torch.nn.Module + config: DeepSpeedInferenceConfig + """ + global DS_INFERENCE_ENABLED + DS_INFERENCE_ENABLED = True + + super().__init__() + if DeepSpeedTransformerInference.workspace is not None: + self.destroy() + + self.module = model + self._config = config + + self._get_model_config_generate(config) # keep for weird backward compatibility + + # patch model generate with ours if model uses it + if hasattr(self.module, "generate"): + self.generate = self._generate + + if hasattr(self.module, "config"): + TransformerPolicy.hf_model_config = self.module.config + + if config.dtype not in get_accelerator().supported_dtypes(): + raise ValueError( + f"Data type {config.dtype} is not supported by {get_accelerator().device_name()} accelerator") + + # todo: keep this self.injection_dict because we don't use to change config.injection_policy API + # todo: this will get changed when Molly's PR on auto injection dict is merged + self.injection_dict = config.injection_policy + + # todo: refactor the mp_group and mp_size related in the next refactor + self.mp_group = config.tensor_parallel.tp_group + self.mpu = config.tensor_parallel.mpu + + self.quantize_merge_count = 1 + self.quantization_scales = None + + # these are not needed in the config as we are creating them ourselves in the inference engine + self.ep_group = None # config.moe.ep_group + self.expert_mp_group = None # config.moe.ep_mp_group + + self.cuda_graph_created = False + self.checkpoint_engine = TorchCheckpointEngine() + quantization_setting = None + self._init_quantization_setting( + quantization_setting) # todo: update with the new quant config for weight quant + self.model_profile_enabled = False + self._model_times = [] + + if not self.injection_dict and config.replace_with_kernel_inject: + # This is a hack to remove the prepare_mask function on HF side for BLOOM architecture + self.remove_mask_prepare_for_bloom() + + if self.injection_dict or not config.replace_with_kernel_inject: + # This is a hack to redefine the alibi func due to TP + if config.tensor_parallel.tp_size > 1: + self.build_alibi_tensor() + self.build_attn_bias() + + if get_accelerator().device_name() == 'cuda' and config.enable_cuda_graph: + assert pkg_version.parse(torch.__version__) >= pkg_version.parse("1.10"), \ + "If you want to use cuda graph, please upgrade torch to at least v1.10" + + # convert model to intended dtype + if config.dtype: + self._convert_to_dtype(config) + + if self.mpu: + config.tensor_parallel.tp_size = dist.get_world_size(group=self.mpu.get_model_parallel_group()) + self.mp_group = self.mpu.get_model_parallel_group() + elif config.tensor_parallel.tp_size > 1: + self._create_model_parallel_group(config) + config.tensor_parallel.tp_group = self.mp_group + + if isinstance(self.module, torch.nn.Module): + moe, _ = has_moe_layers(self.module) + else: + moe = False + + if moe and dist.get_world_size() > 1: + self._create_ep_parallel_group(config.moe.moe_experts) + + # We only support three modes: 1) user specified policy for tensor-parallelism, 2) kernel injection (replace_with_kernel_inject), and 3) automatic tensor parallelism if tp_size > 1. + if self.injection_dict: + # 1. User specified Tensor Parallelism + assert not config.replace_with_kernel_inject, "Cannot use both user specified injection policy and kernel injection" + for client_module, injection_policy in self.injection_dict.items(): + + assert issubclass(client_module, + torch.nn.Module), f"{client_module} is not a subclass of torch.nn.Module" + + # construct the tuple and pass that instead of a string or dict. + if isinstance(injection_policy, str): + config.injection_policy_tuple = (injection_policy, ) + else: + config.injection_policy_tuple = injection_policy + + layer_names = [name for name, _ in self.module.named_modules()] + for policy in config.injection_policy_tuple: + if not any(name.endswith(policy) for name in layer_names): + raise ValueError(f"Injection policy layer'{policy}' not valid.") + + self._apply_injection_policy(config, client_module) + else: + if config.replace_with_kernel_inject: + # 2. DeepSpeed Kernel Injection + self._apply_injection_policy(config) + elif config.tensor_parallel.tp_size > 1: + # 3. Automatic Tensor Parallelism + parser_dict = AutoTP.tp_parser(model) + print("AutoTP: ", parser_dict) + for client_module, injection_policy in parser_dict: + if isinstance(injection_policy, str): + config.injection_policy_tuple = (injection_policy, ) + else: + config.injection_policy_tuple = injection_policy + self._apply_injection_policy(config, client_module) + + device = get_accelerator().current_device_name() + # NOTE: This check assumes a Hugging Face hierarchy for the device type i.e. module.device.type + is_meta_device = hasattr(self.module, "device") and self.module.device.type == 'meta' + if is_meta_device: + self.module.to_empty(device=device) + elif not config.keep_module_on_host: + self.module.to(device) + + if config.tensor_parallel.tp_size > 1: + _rng_state = get_accelerator().get_rng_state().to(get_accelerator().current_device_name()) + dist.broadcast(_rng_state, 0) + get_accelerator().set_rng_state(_rng_state.cpu()) + + if config.tensor_parallel.tp_size > 1: + assert not config.enable_cuda_graph, "Cuda graph is not supported for model parallelism" + + # Check if local CUDA graphs can be created in replacement modules + self.local_cuda_graph = self._local_cuda_graph_used(self.module) + self._is_compiled = False + + def destroy(self): + DeepSpeedTransformerInference.layer_id = 0 + DeepSpeedSelfAttention.num_layers = 0 + if DeepSpeedTransformerInference.workspace.is_allocated(): + DeepSpeedTransformerInference.workspace.release_workspace() + DeepSpeedTransformerInference.workspace = None + + def profile_model_time(self, use_cuda_events=True): + if not self.model_profile_enabled and not self._config.enable_cuda_graph: + self.module.register_forward_pre_hook(self._pre_forward_hook) + self.module.register_forward_hook(self._post_forward_hook) + self.model_profile_enabled = True + self.use_cuda_events = use_cuda_events + if self.use_cuda_events: + self.timers = SynchronizedWallClockTimer() + + # todo: remove this once all the config dicts are centralized from top level pydantic config + def _get_model_config_generate(self, config): + # this is being passed to replace_transformer_layer(config=self.user_model_config_dict) + self.config = getattr(self.module, 'config', None) if config.config is None else config.config + + def remove_mask_prepare_for_bloom(self): + if hasattr(self.module, 'transformer'): + if hasattr(self.module.transformer, '_prepare_attn_mask'): + self.module.transformer._prepare_attn_mask = lambda attention_mask, *args, **kwargs: attention_mask + + def build_alibi_tensor(self): + if hasattr(self.module, 'transformer'): + if hasattr(self.module.transformer, 'build_alibi_tensor'): + self.module.transformer.build_alibi_tensor = build_bloom_alibi_tensor + if hasattr(self.module.transformer, 'build_mpt_alibi_tensor'): + self.module.transformer.build_mpt_alibi_tensor_orig = self.module.transformer.build_mpt_alibi_tensor + self.module.transformer.__class__.build_mpt_alibi_tensor = build_mpt_alibi_tensor + if hasattr(self.module, 'model'): + if hasattr(self.module.model, 'get_alibi_mask'): + self.module.model.get_alibi_mask_orig = self.module.model.get_alibi_mask + self.module.model.__class__.get_alibi_mask = get_alibi_mask + + def build_attn_bias(self): + if hasattr(self.module, 'transformer'): + if hasattr(self.module.transformer, '_attn_bias'): + self.module.transformer._attn_bias_orig = self.module.transformer._attn_bias + self.module.transformer.__class__._attn_bias = build_mpt_atten_bias_tensor + + def _pre_forward_hook(self, module, *inputs, **kwargs): + if self.use_cuda_events: + self.timers(INFERENCE_MODEL_TIMER).start() + else: + get_accelerator().synchronize() + self._start = time.time() + + def _post_forward_hook(self, module, input, output): + if self.use_cuda_events: + self.timers(INFERENCE_MODEL_TIMER).stop() + elapsed_time = self.timers(INFERENCE_MODEL_TIMER).elapsed(reset=True) + else: + get_accelerator().synchronize() + self._end = time.time() + elapsed_time = (self._end - self._start) * 1e3 # convert seconds to ms + self._model_times.append(elapsed_time) + + def _create_model_parallel_group(self, config): + # Call the init process + if InferenceEngine.inference_mp_group is None: + init_distributed() + local_rank = int(os.getenv('LOCAL_RANK', '0')) + get_accelerator().set_device(local_rank) + + ranks = [i for i in range(config.tensor_parallel.tp_size)] + self.mp_group = dist.new_group(ranks) + InferenceEngine.inference_mp_group = self.mp_group + else: + self.mp_group = InferenceEngine.inference_mp_group + + def _create_ep_parallel_group(self, moe_experts): + # Call the init process + self.ep_group = {} + self.expert_mp_group = {} + moe_experts = moe_experts if type(moe_experts) is list else [moe_experts] + for e in moe_experts: + self.ep_group.update({e: None}) + self.expert_mp_group.update({e: None}) + for moe_ep_size in self.ep_group.keys(): + num_ep_groups = dist.get_world_size() // moe_ep_size + for i in range(num_ep_groups): + ep_cnt = i * moe_ep_size + size = dist.get_world_size() if moe_ep_size > dist.get_world_size() else moe_ep_size + ranks = list(range(ep_cnt, ep_cnt + size)) + _ep_group = dist.new_group(ranks) + if dist.get_rank() in ranks: + self.ep_group.update({moe_ep_size: _ep_group}) + + if dist.get_world_size() > moe_ep_size: + num_expert_mp_groups = dist.get_world_size() // num_ep_groups + expert_mp_size = dist.get_world_size() // moe_ep_size + for i in range(num_expert_mp_groups): + expert_mp_comm_ranks = [i + nr * moe_ep_size for nr in range(expert_mp_size)] + _expert_mp_group = dist.new_group(expert_mp_comm_ranks) + if dist.get_rank() in expert_mp_comm_ranks: + self.expert_mp_group.update({moe_ep_size: _expert_mp_group}) + + def _init_quantization_setting(self, quantization_setting): + self.quantize_bits = 8 + self.mlp_extra_grouping = False + self.quantize_groups = 1 + if type(quantization_setting) is tuple: + self.mlp_extra_grouping, \ + self.quantize_groups = quantization_setting + elif quantization_setting is not None: + self.quantize_groups = quantization_setting + log_dist( + f"quantize_bits = {self.quantize_bits} " + f"mlp_extra_grouping = {self.mlp_extra_grouping}, " + f"quantize_groups = {self.quantize_groups}", [0]) + + def load_model_with_checkpoint(self, r_module): + self.mp_replace = ReplaceWithTensorSlicing( + mp_group=self.mp_group, mp_size=self._config.tensor_parallel.tp_size) #, out_dim=0, in_dim=1) + error_msgs = [] + + def load(module, state_dict, prefix): + args = (state_dict, prefix, {}, True, [], [], error_msgs) + if hasattr(module, 'weight'): + if module.weight.data.is_meta: + # meta tensor cannot be casted or copied to, so we need to replace it with a normal tensor here + module.weight = torch.nn.parameter.Parameter(data=torch.empty_like(module.weight.data, + device="cpu"), + requires_grad=module.weight.data.requires_grad) + if 'query_key_value' in prefix: + module.weight = self.mp_replace.strided_copy(module.weight.data, + state_dict[prefix + 'weight'], + num_splits=3) + else: + module.weight = self.mp_replace.copy(module.weight.data, state_dict[prefix + 'weight']) + else: + if module.norm.weight.data.is_meta: + # meta tensor cannot be casted or copied to, so we need to replace it with a normal tensor here + module.norm.weight = torch.nn.parameter.Parameter( + data=torch.empty_like(module.norm.weight.data, device="cpu"), + requires_grad=module.norm.weight.data.requires_grad) + module.norm.weight = self.mp_replace.copy(module.norm.weight.data, state_dict[prefix + 'weight']) + if prefix + 'bias' in self.key_list: + if hasattr(module, 'norm'): + if module.norm.bias.data.is_meta: + # meta tensor cannot be casted or copied to, so we need to replace it with a normal tensor here + module.norm.bias = torch.nn.parameter.Parameter( + data=torch.empty_like(module.norm.bias.data, device="cpu"), + requires_grad=module.norm.bias.data.requires_grad) + module.norm.bias = self.mp_replace.copy(module.norm.bias, state_dict[prefix + 'bias']) + else: + if module.bias.data.is_meta: + # meta tensor cannot be casted or copied to, so we need to replace it with a normal tensor here + module.bias = torch.nn.parameter.Parameter(data=torch.empty_like(module.bias.data, + device="cpu"), + requires_grad=module.bias.data.requires_grad) + data = state_dict[prefix + 'bias'] + data = data.to(get_accelerator().current_device_name()) + module.bias = self.mp_replace.copy(module.bias, data) + + layer_policies = { + nn.Linear: load, + nn.Embedding: load, + nn.LayerNorm: load, + LinearLayer: load, + LinearAllreduce: load + } + + def load_module_recursive(module, prefix='', level=0): + for name, child in module.named_children(): + if child.__class__ in layer_policies: + checking_key = prefix + name + '.' + if not any(checking_key in item for item in self.key_list): + continue + if len(list(child.parameters())) > 0 and list(child.parameters())[0].numel() == 0: + if len(child.weight.ds_shape) == 1: + child = Normalize(dim=child.weight.ds_shape[-1], dtype=child.weight.dtype, eps=child.eps) + setattr(module, name, child) + load(child, self.sd, prefix + name + '.') + else: + load_module_recursive(child, prefix if level == 0 else prefix + name + '.', level + 1) + + load_module_recursive(r_module) + + embedding_weight = None + + for n, p in r_module.named_parameters(): + if "word_embeddings." in n or "embed_tokens." in n or "wte." in n: + embedding_weight = p + if embedding_weight is not None and hasattr(r_module, "lm_head") and hasattr( + r_module.lm_head, "weight") and r_module.lm_head.weight.is_meta: + r_module.lm_head.weight = embedding_weight + + def _apply_injection_policy(self, config, client_module=None): + # client_module is only passed when using the injection_dict method. + checkpoint_dir = config.checkpoint + checkpoint = SDLoaderFactory.get_sd_loader_json(checkpoint_dir, + self.checkpoint_engine) if checkpoint_dir is not None else None + + generic_injection(self.module, dtype=config.dtype, enable_cuda_graph=config.enable_cuda_graph) + + if isinstance(self.module, torch.nn.Module): + # config is our DeepSpeedInferenceConfig and self.config is the HF model config + replace_transformer_layer(client_module, self.module, checkpoint, config, self.config) + + def _get_all_ckpt_names(self, checkpoints_path, tag): + ckpt_file_pattern = self._get_ckpt_name(checkpoints_path, tag, mp_placeholder="*") + import glob + + ckpt_files = glob.glob(ckpt_file_pattern) + ckpt_files.sort() + return ckpt_files + + def _get_ckpt_name(self, checkpoints_path, tag, mp_placeholder=None): + if mp_placeholder is not None: + mp_rank_str = mp_placeholder + else: + mp_rank = 0 if self.mpu is None else self.mpu.get_model_parallel_rank() + mp_rank_str = "{:02d}".format(mp_rank) + + ckpt_name = os.path.join( + checkpoints_path, + "mp_rank_" + mp_rank_str + "_model_states.pt", + ) + return ckpt_name + + def _load_checkpoint(self, load_dir, load_module_strict=True, tag=None): + is_pipe_parallel = isinstance(self.module, PipelineModule) + if is_pipe_parallel: + raise RuntimeError('pipeline parallelism is currently not supported in inference.') + if not isinstance(load_dir, dict) and os.path.isdir(load_dir): + if tag is None: + latest_path = os.path.join(load_dir, "latest") + if os.path.isfile(latest_path): + with open(latest_path, "r") as fd: + tag = fd.read().strip() + + ckpt_list = self._get_all_ckpt_names(load_dir, tag) + sd_loader = SDLoaderFactory.get_sd_loader(ckpt_list, self.checkpoint_engine) + else: + sd_loader = SDLoaderFactory.get_sd_loader_json(load_dir, self.checkpoint_engine) + + checkpoint = sd_loader['checkpoints'] + + if type(checkpoint) is list: + self.sd = torch.load(checkpoint[0], map_location='cpu', weights_only=False) + self.key_list = list(self.sd.keys()) + + self.load_model_with_checkpoint(self.module) + + for i in range(1, len(checkpoint)): + if not dist.is_initialized() or dist.get_rank() == 0: + print(f"loading checkpoint ({i})") + self.sd = torch.load(checkpoint[i], map_location=get_accelerator().device_name(), weights_only=False) + self.key_list = list(self.sd.keys()) + self.load_model_with_checkpoint(self.module) + else: + mp_rank = 0 if self.mpu is None else self.mpu.get_model_parallel_rank() + + load_path, checkpoint, quantize_config = sd_loader.load(self._config.tensor_parallel.tp_size, + mp_rank, + is_pipe_parallel=is_pipe_parallel, + quantize=(self._config.dtype is torch.int8), + quantize_groups=self.quantize_groups, + mlp_extra_grouping=self.mlp_extra_grouping) + + self.quantization_scales, self.quantize_merge_count = quantize_config + + moe, _ = has_moe_layers(self.module) + if moe: + from deepspeed.runtime.engine import DeepSpeedEngine + old_moe_load = False + if not isinstance(checkpoint['num_experts'], list): + old_moe_load = True + DeepSpeedEngine.load_moe_state_dict(load_dir, + tag, + state_dict=checkpoint[self._choose_module_key(checkpoint)], + old_moe_load=old_moe_load, + model=self.module, + mpu=self.mpu, + checkpoint_engine=self.checkpoint_engine) + + self.module.load_state_dict(state_dict=checkpoint[self._choose_module_key(checkpoint)], + strict=load_module_strict) + + def _choose_module_key(self, sd): + assert not ('module' in sd + and 'model' in sd), "checkpoint has both 'model' and 'module' keys, not sure how to proceed" + assert 'module' in sd or 'model' in sd, "checkpoint contains neither 'model' or 'module' keys, not sure how to proceed" + if 'module' in sd: + return 'module' + elif 'model' in sd: + return 'model' + + def _convert_to_dtype(self, config): + if not isinstance(self.module, torch.nn.Module): + return + + if False: #config.dtype is torch.int8 and self.quantization_scales is None: + quantizer = WeightQuantization(mlp_extra_grouping=self.mlp_extra_grouping) + model, self.quantization_scales = quantizer.model_quantize(self.module, self.injection_dict, + self.quantize_bits, self.quantize_groups) + elif config.dtype == torch.half: + self.module.half() + elif config.dtype == torch.bfloat16: + self.module.bfloat16() + elif config.dtype == torch.float: + self.module.float() + + def _create_cuda_graph(self, *inputs, **kwargs): + # warmup to create the workspace and cublas handle + cuda_stream = get_accelerator().Stream() + cuda_stream.wait_stream(get_accelerator().current_stream()) + with get_accelerator().stream(cuda_stream): + for i in range(3): + ret = self.module(*inputs, **kwargs) + get_accelerator().current_stream().wait_stream(cuda_stream) + + # create cuda_graph and assign static_inputs and static_outputs + self._cuda_graphs = get_accelerator().create_graph() + self.static_inputs = inputs + self.static_kwargs = kwargs + + with get_accelerator().capture_to_graph(self._cuda_graphs): + self.static_output = self.module(*self.static_inputs, **self.static_kwargs) + + self.cuda_graph_created = True + + def _graph_replay(self, *inputs, **kwargs): + for i in range(len(inputs)): + if torch.is_tensor(inputs[i]): + self.static_inputs[i].copy_(inputs[i]) + for k in kwargs: + if torch.is_tensor(kwargs[k]): + self.static_kwargs[k].copy_(kwargs[k]) + get_accelerator().replay_graph(self._cuda_graphs) + return self.static_output + + def model_times(self): + assert self.model_profile_enabled, "model profiling is not enabled" + model_times = self._model_times + if self._config.enable_cuda_graph and len(self._model_times) == 0: + raise ValueError("Model times are empty and cuda graph is enabled. If " + "this is a GPT-style model this combo is not supported. If this is a " + "BERT-style model this is a bug, please report it. " + f"Model type is: {type(self.module)}") + self._model_times = [] + return model_times + + def _module_match(self, module): + for policy in generic_policies: + policy = policy() + if policy.match_replaced(module): + return True + return False + + def _local_cuda_graph_used(self, module): + if isinstance(module, torch.nn.Module): + return False + else: + sub_module_cuda_graph = False + for name in module.__dict__.keys(): + sub_module = getattr(module, name) + + if self._module_match(sub_module) and hasattr(sub_module, "enable_cuda_graph"): + sub_module_cuda_graph = True + + return sub_module_cuda_graph + + def forward(self, *inputs, **kwargs): + """Execute forward propagation + + Arguments: + *inputs: Variable length input list + **kwargs: variable length keyword arguments + """ + start = None + if self.model_profile_enabled and get_accelerator().device_name() == 'cuda' and self._config.enable_cuda_graph: + get_accelerator().synchronize() + start = time.time() + + if get_accelerator().device_name() == 'cuda' and self._config.enable_cuda_graph and not self.local_cuda_graph: + if self.cuda_graph_created: + outputs = self._graph_replay(*inputs, **kwargs) + else: + self._create_cuda_graph(*inputs, **kwargs) + outputs = self._graph_replay(*inputs, **kwargs) + + else: + outputs = self.module(*inputs, **kwargs) + + if self.model_profile_enabled and self._config.enable_cuda_graph: + get_accelerator().synchronize() + duration = (time.time() - start) * 1e3 # convert seconds to ms + self._model_times.append(duration) + + return outputs + + def _generate(self, *inputs, **kwargs): + # Reset KV-cache at the beginning of generate + if hasattr(self.module, 'reset_cache'): + self.module.reset_cache() + num_beams = 1 + if "generation_config" in kwargs: + gen_config = kwargs["generation_config"] + num_beams = getattr(gen_config, "num_beams", 1) + if "num_beams" in kwargs: + num_beams = kwargs["num_beams"] + + if num_beams > 1: + raise NotImplementedError("DeepSpeed does not support `num_beams` > 1, if this is important to you please " + "add your request to: https://github.com/deepspeedai/DeepSpeed/issues/2506") + + if ("input_ids" in kwargs) and (kwargs["input_ids"].dim() == 2): + for input_tensor in kwargs["input_ids"]: + tensor_length = input_tensor.shape[-1] + if tensor_length > self._config.max_out_tokens: + raise RuntimeError( + f"Input with size {tensor_length} exceeds maximum length of {self._config.max_out_tokens}. Please increase max_tokens in the DeepSpeed Inference Config." + ) + + return self.module.generate(*inputs, **kwargs) + + def compile(self, backend=get_accelerator().get_compile_backend(), compile_kwargs={}) -> None: + """ + Compile the module using the specified backend and kwargs. + """ + if not is_compile_supported(): + raise RuntimeError("compile is not supported in your version of PyTorch.") + + if self._is_compiled: + return + + # Avoid graph breaks + deepspeed.utils.nvtx.enable_nvtx = False + self.module.compile(backend=backend, **compile_kwargs) + self._is_compiled = True + + @property + def is_compiled(self) -> bool: + return self._is_compiled diff --git a/lib/python3.12/site-packages/deepspeed/inference/quantization/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/quantization/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..208299fb8c50f73468d293b6fa5dca71649d62e7 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/quantization/__init__.py @@ -0,0 +1,4 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team diff --git a/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a3c8d1ce8cd42a8af50156285c796e6d7b2f85de Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/layers.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/layers.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ebb0d18aacfb0fe2826b463b370ad9cf12b753a5 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/layers.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/quantization.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/quantization.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..172fce36ce611c3b6fd0eb4a1a88fb9fcce9a4d6 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/quantization.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/quantization_context.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/quantization_context.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..deacda2693d56501998c1ebed38edfe8c01068b3 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/quantization_context.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/utils.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/utils.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..cf6e58fa2aed988b9895fb5f7df28c9889b3ff15 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/quantization/__pycache__/utils.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/quantization/layers.py b/lib/python3.12/site-packages/deepspeed/inference/quantization/layers.py new file mode 100644 index 0000000000000000000000000000000000000000..e9a7e5629f1b2f3c4cd57889f3276e876cecc7db --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/quantization/layers.py @@ -0,0 +1,114 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from torch import nn +from torch import Tensor +from torch.nn import functional as F +from .utils import Quantizer, DeQuantizer, concat_to_compat_param +from typing import Tuple, Callable, Dict +from deepspeed.runtime.zero import register_external_parameter + +quantized_weight_registry = {} +is_zero3_enabled = False + + +# deal with weight sharing +def get_quantized_weight_wrapper(model, pre_quant_weight: nn.Parameter, quantize_weight_fn: Callable) -> nn.Parameter: + if id(pre_quant_weight) in quantized_weight_registry: + compat_tensor = quantized_weight_registry[id(pre_quant_weight)] + if is_zero3_enabled: + register_external_parameter(model, compat_tensor) + + return quantized_weight_registry[id(pre_quant_weight)] + else: + quantized_weights, quant_scale, quant_min = quantize_weight_fn() + quantized_weight_registry[id(pre_quant_weight)] = concat_to_compat_param(quantized_weights, quant_scale, + quant_min) + return quantized_weight_registry[id(pre_quant_weight)] + + +def get_quantize_weight_fn(quantizer: Quantizer, pre_quant_weight: nn.Parameter) -> Callable: + + def func() -> Tuple[nn.Parameter, Tensor, Tensor]: + quantized_weights, quant_scale, quant_min = quantizer.quantize(pre_quant_weight.data) + # A temporary hack as zero Zero3 assume all model weights has the same type. in all_gather_coalesced.get_only_unique_item + quantized_weights = quantized_weights.view(pre_quant_weight.dtype) + quant_scale = quant_scale.type(pre_quant_weight.dtype) + quant_min = quant_min.type(pre_quant_weight.dtype) + return quantized_weights, quant_scale, quant_min + + return func + + +class QuantizedLinear(nn.Linear): + + def __init__(self, config: Dict, pre_quant_layer: nn.Linear) -> None: + super(QuantizedLinear, self).__init__(in_features=pre_quant_layer.in_features, + out_features=pre_quant_layer.out_features, + bias=pre_quant_layer.bias is not None, + device=pre_quant_layer.weight.device, + dtype=pre_quant_layer.weight.dtype) + self.config = config + + self.quantizer = Quantizer(config=config) + self.bias = pre_quant_layer.bias + self.weight = get_quantized_weight_wrapper(self, pre_quant_layer.weight, + get_quantize_weight_fn(self.quantizer, pre_quant_layer.weight)) + + self.weight.dequantizer = DeQuantizer(config, pre_quant_layer.weight.dtype) + + def forward(self, input: Tensor) -> Tensor: + quantized_weight, quant_scale, quant_min = self.weight.deconcat(self.weight) + temp_dequantized_weight = self.weight.dequantizer.dequantize(quantized_weight.view(torch.uint8), quant_scale, + quant_min) + + # !!! Do not use torch.functional.linear(input, temp_dequantized_weight, self.bias) here as in zero3 torch.functional.linear is + # replaced by LinearFunctionForZeroStage3. Which assume weight is non-temporary. + # If weight is temp buffer there will be memory leak. + return torch._C._nn.linear(input, temp_dequantized_weight, self.bias) + + +class QuantizedEmbedding(nn.Embedding): + + def __init__(self, config: Dict, pre_quant_layer: nn.Embedding) -> None: + super(QuantizedEmbedding, self).__init__(num_embeddings=pre_quant_layer.num_embeddings, + embedding_dim=pre_quant_layer.embedding_dim, + padding_idx=pre_quant_layer.padding_idx, + max_norm=pre_quant_layer.max_norm, + norm_type=pre_quant_layer.norm_type, + scale_grad_by_freq=pre_quant_layer.scale_grad_by_freq, + sparse=pre_quant_layer.sparse, + _weight=pre_quant_layer.weight, + device=pre_quant_layer.weight.device, + dtype=pre_quant_layer.weight.dtype) + + assert pre_quant_layer.max_norm is None, 'Not supported' + assert pre_quant_layer.norm_type == 2, 'Not supported' + assert pre_quant_layer.scale_grad_by_freq == False, 'Not supported' + assert pre_quant_layer.sparse == False, 'Not supported' + + self.config = config + quantizer = Quantizer(config=config) + + self.weight = get_quantized_weight_wrapper(self, pre_quant_layer.weight, + get_quantize_weight_fn(quantizer, pre_quant_layer.weight)) + + self.weight.dequantizer = DeQuantizer(config, pre_quant_layer.weight.dtype) + + def forward(self, input: Tensor) -> Tensor: + quantized_weight, quant_scale, quant_min = self.weight.deconcat(self.weight) + temp_dequantized_weight = self.weight.dequantizer.dequantize(quantized_weight.view(torch.uint8), quant_scale, + quant_min) + + return F.embedding(input, temp_dequantized_weight, self.padding_idx, self.max_norm, self.norm_type, + self.scale_grad_by_freq, self.sparse) + + +QUANTIZATION_LAYER_MAPPINGS = { + nn.Linear: QuantizedLinear, + nn.Embedding: QuantizedEmbedding, +} diff --git a/lib/python3.12/site-packages/deepspeed/inference/quantization/quantization.py b/lib/python3.12/site-packages/deepspeed/inference/quantization/quantization.py new file mode 100644 index 0000000000000000000000000000000000000000..9ae39e8d568839f12d14aa98569677b5de9a7086 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/quantization/quantization.py @@ -0,0 +1,111 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch +from torch import nn +from typing import Dict +import gc +from deepspeed.inference.quantization import layers +from .layers import QUANTIZATION_LAYER_MAPPINGS +from .utils import get_AsyncPartitionedParameterSwapper, recursive_setattr +from deepspeed.utils.logging import logger +from collections import deque +from transformers.utils.generic import ContextManagers +from .quantization_context import QuantizationContext +import contextlib + + +def _init_group_wise_weight_quantization(model: nn.Module, ds_config: Dict) -> nn.Module: + """[Experimental] Apply group-wise weight quantization to model. Replace layers module according to config_list + + Args: + model (nn.Module): A nn.Module + ds_config (Dict, optional): The ds_config dictionary. use None for non-deepspeed managed model. + + Returns: + nn.Module: Quantized nn.Module + """ + + # global quantized_weight_registry + + matched_module_list_by_key = {} + matched_module_count = 0 + + assert 'weight_quantization' in ds_config, 'Please provide quantization config in ds_config' + quantization_config = ds_config['weight_quantization']['post_init_quant'] + + # Return nvme swapper if exists, else return None. + # For nvme offloading we must use the same swapper here as model initialized. + nvme_swapper = get_AsyncPartitionedParameterSwapper(model) + is_zero3_enabled = 'zero_optimization' in ds_config and \ + 'stage' in ds_config['zero_optimization'] and \ + ds_config['zero_optimization']['stage'] == 3 + is_offloading_enabled = 'zero_optimization' in ds_config and \ + 'offload_param' in ds_config['zero_optimization'] + + layers.is_zero3_enabled = is_zero3_enabled + + context_mgr = ContextManagers([QuantizationContext(config_dict_or_path=ds_config, param_swapper=nvme_swapper)]) \ + if is_zero3_enabled else contextlib.suppress() + with context_mgr: + module_list = list( + filter(lambda named_module: type(named_module[1]) in QUANTIZATION_LAYER_MAPPINGS, model.named_modules())) + + # Quantize small weight first then large. + if not is_offloading_enabled: + module_list.sort(key=lambda named_module: named_module[1].weight.ds_tensor.numel() + if is_zero3_enabled else named_module[1].weight.numel()) + module_list = deque(module_list) + + while len(module_list) > 0: + # Use popleft to timely release module's memory of replaced module after each loop iteration + module_name, module = module_list.popleft() + + matched_key = None + matched_quantization_config = None + + for key, config in quantization_config.items(): + if key in module_name: + assert matched_key is None, f'{module_name} matched multiple quantization key word {matched_key} and {key}' + matched_key = key + matched_quantization_config = config + + if matched_key is None: + continue + + if is_zero3_enabled: + module.weight.all_gather() + + assert module.weight.dtype == torch.float16, 'Model weight is expected in half.' + + new_module = QUANTIZATION_LAYER_MAPPINGS[type(module)](matched_quantization_config, module) + + if is_zero3_enabled: + module.weight.partition() + + recursive_setattr(model, module_name, new_module) + + if matched_key not in matched_module_list_by_key: + matched_module_list_by_key[matched_key] = [] + matched_module_list_by_key[matched_key].append(module_name) + matched_module_count += 1 + + # Timely recycle memory to prevent OOM on large models + gc.collect() + + # Clear registry after model construction. + layers.quantized_weight_registry.clear() + + logger.info( + f'Group-wise weight quantization summary: convert {matched_module_count} node(s) to quantized implementation') + summary_str = '\n' + + for key, module_list in matched_module_list_by_key.items(): + summary_str += f'Key: {key}, matched modules:\n' + for module_name in module_list: + summary_str += f'\t{module_name}\n' + logger.info(summary_str) + + return model diff --git a/lib/python3.12/site-packages/deepspeed/inference/quantization/quantization_context.py b/lib/python3.12/site-packages/deepspeed/inference/quantization/quantization_context.py new file mode 100644 index 0000000000000000000000000000000000000000..d3333da0505883f032b18bc356a636a7b88170a8 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/quantization/quantization_context.py @@ -0,0 +1,13 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from deepspeed.runtime.zero import partition_parameters +from deepspeed.runtime.swap_tensor.partitioned_param_swapper import AsyncPartitionedParameterSwapper + + +class QuantizationContext(partition_parameters.Init): + + def __init__(self, config_dict_or_path, param_swapper: AsyncPartitionedParameterSwapper = None) -> None: + super().__init__(config_dict_or_path=config_dict_or_path, param_swapper=param_swapper) diff --git a/lib/python3.12/site-packages/deepspeed/inference/quantization/utils.py b/lib/python3.12/site-packages/deepspeed/inference/quantization/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..a5e8f28bdec91b5ce6ce1c9d97d2f0199ed8b713 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/quantization/utils.py @@ -0,0 +1,288 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch +import deepspeed +from torch import Tensor +from typing import Tuple +import torch.nn as nn +from typing import Dict, Callable, Union +from deepspeed.accelerator import get_accelerator +import functools + +device = get_accelerator().device_name() if get_accelerator().is_available() else 'cpu' + +quantizer_module = None + + +def get_quantizer_module(): + global quantizer_module + if quantizer_module is None: + quantizer_module = deepspeed.ops.op_builder.QuantizerBuilder().load() + return quantizer_module + + +def tensor_clamp(tensor: Tensor, min, max) -> Tensor: + if tensor.device.type == 'cpu' and tensor.dtype == torch.float16: + # CPU does not support FP16 clamp + return tensor.to(dtype=torch.float32).clamp_(min, max).to(dtype=torch.float16) + else: + return tensor.clamp_(min, max) + + +def tensor_round(tensor: Tensor) -> Tensor: + if tensor.device.type == 'cpu' and tensor.dtype == torch.float16: + # CPU does not support FP16 round + return tensor.to(dtype=torch.float32).round_().to(dtype=torch.float16) + else: + return tensor.round_() + + +class Quantizer: + + def __init__(self, config: Dict) -> None: + self.config = config + assert self.config['num_bits'] == 4 or self.config[ + 'num_bits'] == 8, 'Only INT4 and INT8 quantization is supported.' + assert self.config['symmetric'] == False, 'Only asymmetric quantization is supported at this moment.' + + def quantize(self, tensor: Tensor) -> Tuple[Tensor, Tensor, Tensor]: + assert tensor.shape[self.config['group_dim']] % self.config['group_size'] == 0 \ + , f'Tensor shape: {tensor.shape} quantization config {self.config}' + + tensor = torch.clone(tensor) + + shape = tensor.shape + num_groups = shape[self.config['group_dim']] // self.config['group_size'] + new_shape = (shape[:self.config['group_dim']] + (num_groups, self.config['group_size']) + + shape[self.config['group_dim'] + 1:]) + tensor = tensor.view(new_shape) + + quantized_tensor, scale, min_value = self._quantize_int8(tensor) + quantized_tensor = quantized_tensor.view(shape) + + if self.config['num_bits'] == 4: + return self._compress_uint8_to_uint4(quantized_tensor), scale, min_value + if self.config['num_bits'] == 8: + return quantized_tensor, scale, min_value + + assert False, 'Unsupported quantization bits {}'.format(self.config['num_bits']) + + def _quantize_int8(self, tensor: Tensor) -> Tuple[Tensor, Tensor, Tensor]: + q_range = 2**self.config['num_bits'] - 1 + min_value = tensor.amin(dim=self.config['group_dim'] + 1, keepdim=True) + max_value = tensor.amax(dim=self.config['group_dim'] + 1, keepdim=True) + + scale = q_range / (max_value - min_value) + + tensor = tensor.sub_(min_value).mul_(scale) + tensor = tensor_round(tensor_clamp(tensor, 0, q_range)).to(torch.uint8) + return tensor, scale, min_value + + def _compress_uint8_to_uint4(self, tensor: Tensor) -> Tensor: + assert tensor.shape[-1] % 2 == 0 + + new_data_shape = list(tensor.shape) + new_data_shape[-1] = new_data_shape[-1] // 2 + + data = torch.empty(new_data_shape, dtype=torch.uint8, device=tensor.device) + data = torch.bitwise_or(tensor[..., 0::2].bitwise_left_shift(4), tensor[..., 1::2]) + + return data + + +class DeQuantizer: + + def __init__(self, config: Dict, dtype: torch.dtype) -> None: + self.config = config + self.dtype = dtype + assert self.config['num_bits'] == 4 or self.config[ + 'num_bits'] == 8, 'Only INT4 and INT8 quantization is supported.' + assert self.config['symmetric'] == False, 'Only asymmetric quantization is supported at this moment.' + + def dequantize(self, tensor: Tensor, quant_scale: Tensor, quant_min: Tensor) -> Tensor: + # Use customized CUDA quantization kernel if possible. + if self.config['group_size'] % 8 == 0 and \ + (self.config['num_bits'] == 4 or self.config['num_bits'] == 8) and \ + self.config['group_dim'] == len(tensor.shape) - 1 and \ + self.dtype == torch.float16 and device == get_accelerator().device_name(): + + last_dimension_size = self.config['group_size'] + if self.config['num_bits'] == 4: + last_dimension_size = last_dimension_size // 2 + quantized_tensor = get_quantizer_module().dequantize_int4_to_half_experimental( + tensor.reshape(-1, last_dimension_size), quant_scale, quant_min, + tensor.numel() // last_dimension_size, self.config['group_size']) + shape = list(tensor.shape) + shape[-1] = shape[-1] * 2 + elif self.config['num_bits'] == 8: + # last_dimension_size = last_dimension_size // 2 + quantized_tensor = get_quantizer_module().dequantize_int8_to_half_experimental( + tensor.reshape(-1, last_dimension_size), quant_scale, quant_min, + tensor.numel() // last_dimension_size, self.config['group_size']) + shape = list(tensor.shape) + + return quantized_tensor.reshape(shape) + + if self.config['num_bits'] == 4: + tensor = self._decompress_uint4_to_uint8(tensor) + elif self.config['num_bits'] != 8: + assert False, 'Unsupported quantization bits {}'.format(self.config['num_bits']) + + shape = tensor.shape + num_groups = shape[self.config['group_dim']] // self.config['group_size'] + new_shape = (shape[:self.config['group_dim']] + (num_groups, self.config['group_size']) + + shape[self.config['group_dim'] + 1:]) + tensor = tensor.view(new_shape) + + dequantized_tensor = self._dequantize_int8(tensor, quant_scale, quant_min).view(shape) + return dequantized_tensor + + def _dequantize_int8(self, tensor: Tensor, quant_scale: Tensor, quant_min: Tensor) -> Tensor: + assert tensor.dtype == torch.uint8 + data = torch.zeros_like(tensor, dtype=self.dtype, device=tensor.device) + data = data.copy_(tensor) + data = data.div_(quant_scale).add_(quant_min) + + return data + + def _decompress_uint4_to_uint8(self, tensor: Tensor) -> Tensor: + new_data_shape = list(tensor.shape) + new_data_shape[-1] = new_data_shape[-1] * 2 + data = torch.empty(new_data_shape, dtype=torch.uint8, device=tensor.device) + data[..., 0::2] = tensor.bitwise_right_shift(4) + data[..., 1::2] = tensor.bitwise_and(0xF) + + return data + + +def get_AsyncPartitionedParameterSwapper(model: nn.Module): + for param_name, param in model.named_parameters(): + if hasattr(param, 'nvme_swapper') and param.nvme_swapper is not None: + return param.nvme_swapper + return None + + +def recursive_setattr(model, module_name, module): + """ + Recursively set the attribute of a module. + Args: + model (`torch.nn.Module`) + The model to set the attribute in. + module_name (`str`) + The name of the module to set the attribute in. + module (`torch.nn.Module`) + The module to set the attribute to. + """ + split_list = module_name.split('.') + output = model + for name in split_list[:-1]: + output = getattr(output, name) + output.__setattr__(split_list[-1], module) + + +def concat_to_compat_param(quantized_weight: Tensor, + quant_scale: Tensor, + quant_min: Tensor, + return_param: bool = True) -> Union[nn.Parameter, Tensor]: + shape_wieght = quantized_weight.shape + shape_scale = quant_scale.shape + shape_min = quant_min.shape + + quantized_weight = torch.flatten(quantized_weight) + quant_scale = torch.flatten(quant_scale) + quant_min = torch.flatten(quant_min) + + def deconcat_individual_tensors(shape_wieght: torch.Size, shape_scale: torch.Size, + shape_min: torch.Size) -> Callable: + + def fn(compat_tensor: nn.Parameter) -> Tuple[Tensor, Tensor, Tensor]: + weight = torch.narrow(compat_tensor, 0, 0, shape_wieght.numel()).view(shape_wieght) + scale = torch.narrow(compat_tensor, 0, shape_wieght.numel(), shape_scale.numel()).view(shape_scale) + min_val = torch.narrow(compat_tensor, 0, + shape_wieght.numel() + shape_scale.numel(), shape_min.numel()).view(shape_min) + + return weight, scale, min_val + + return fn + + compat_tensor = torch.concat([quantized_weight, quant_scale, quant_min]) + if return_param: + compat_tensor = nn.Parameter(compat_tensor, requires_grad=False) + compat_tensor.deconcat = deconcat_individual_tensors(shape_wieght, shape_scale, shape_min) + + return compat_tensor + + +def _quantize_param(param: nn.Parameter, quant_config: Dict): + assert not hasattr(param, 'weight_quantized'), 'Parameter has already been quantized.' + quantizer = Quantizer(quant_config) + dequantizer = DeQuantizer(quant_config, param.dtype) + + quantized_weight, quant_scale, quant_min = quantizer.quantize(param.data) + + quantized_weight = quantized_weight.view(param.dtype) + quant_scale = quant_scale.view(param.dtype) + quant_min = quant_min.view(param.dtype) + + quantized_compat_tensor = concat_to_compat_param(quantized_weight, quant_scale, quant_min) + param.data = quantized_compat_tensor + param.deconcat = quantized_compat_tensor.deconcat + + param.quantizer = quantizer + param.dequantizer = dequantizer + setattr(param, 'weight_quantized', True) + + +def wrap_quantized_functional(f): + + @functools.wraps(f) + def wrapper(input: Tensor, weight: nn.Parameter, *args, **kwargs) -> Tensor: + if hasattr(weight, 'weight_quantized') and getattr(weight, 'weight_quantized'): + quantized_weight, quant_scale, quant_min = weight.deconcat(weight) + temp_dequantized_weight = weight.dequantizer.dequantize(quantized_weight.view(torch.uint8), quant_scale, + quant_min) + return f(input, temp_dequantized_weight, *args, **kwargs) + else: + return f(input, weight, *args, **kwargs) + + return wrapper + + +def wrap_load_from_state_dict(f): + + @functools.wraps(f) + def wrapper(model, state_dict, prefix, local_metadata, strict, missing_keys, unexpected_keys, error_msgs): + replaced_old_value = None + key = None + # We may have nested wrappers if we launch multiple initialization context. + # Use state_dict_quantized flag to quantize state_dict only once + if hasattr(model.weight, 'weight_quantized') and getattr( + model.weight, 'weight_quantized') and not hasattr(model.weight, 'state_dict_quantized'): + setattr(model.weight, 'state_dict_quantized', True) + key = prefix + 'weight' + if key in state_dict: + quantized_weight, quant_scale, quant_min = model.weight.quantizer.quantize(state_dict[key]) + quantized_weight = quantized_weight.view(model.weight.dtype) + quant_scale = quant_scale.view(model.weight.dtype) + quant_min = quant_min.view(model.weight.dtype) + + replaced_old_value = state_dict[key] + + state_dict[key] = concat_to_compat_param(quantized_weight, quant_scale, quant_min) + + f(model, state_dict, prefix, local_metadata, strict, missing_keys, unexpected_keys, error_msgs) + + if replaced_old_value is not None: + state_dict[key] = replaced_old_value + delattr(model.weight, 'state_dict_quantized') + + return wrapper + + +WEIGHT_QUANTIZATION_LAYERS = ( + nn.Linear, + nn.Embedding, +) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..ac8a42da8ab3d9e4ace2a4f1d7b1d455cf7be7fb --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/__init__.py @@ -0,0 +1,7 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team +from .config_v2 import RaggedInferenceEngineConfig, DeepSpeedTPConfig +from .engine_v2 import InferenceEngineV2 +from .engine_factory import build_hf_engine, build_engine_from_ds_checkpoint diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..9659a93c3a3857524041276adba8ea9e949ed37c Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/allocator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/allocator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..0ea6f42beb2f5266b19b16dcda40007b5d96d47f Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/allocator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/config_v2.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/config_v2.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..5402c62760b983dee2e5759a41ebfda349431d44 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/config_v2.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/engine_v2.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/engine_v2.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c17cc271a5b32081e8ade6283d351c2e75731f5c Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/engine_v2.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/inference_parameter.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/inference_parameter.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..32f052833d5c829a00426d5bda48301519808b71 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/inference_parameter.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/inference_utils.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/inference_utils.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..62f98e64a8c6919971ae414c8c2201e03cfe24b2 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/inference_utils.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/logging.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/logging.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d5e4d207df3da8f9a05823fd56bf10d200915bd3 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/logging.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/scheduling_utils.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/scheduling_utils.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..70dcf634f05d0dcfe8cb7c5de56349524564ce56 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/__pycache__/scheduling_utils.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/allocator.py b/lib/python3.12/site-packages/deepspeed/inference/v2/allocator.py new file mode 100644 index 0000000000000000000000000000000000000000..fcc0d94c0f825170dd89c54db73e53a2baf0c077 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/allocator.py @@ -0,0 +1,42 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from functools import reduce +from typing import Iterable +from collections import defaultdict +import torch + +from deepspeed.accelerator import get_accelerator + + +class Allocator: + cache = defaultdict(dict) + + def empty_from(tensor: torch.Tensor, shape: Iterable[int]) -> torch.Tensor: + try: + return Allocator.cache[tensor][shape] + except KeyError: + shape_size = reduce(lambda x, y: x * y, shape) + if shape_size == 0: + raise ValueError("Cannot create empty tensor with size 0") + Allocator.cache[tensor][shape] = tensor.flatten()[:shape_size].view(shape) + return Allocator.cache[tensor][shape] + + +empty_from = Allocator.empty_from + + +def on_device(method) -> torch.Tensor: + """ + Wraps a method to ensure the returned tensor is on the current device. + """ + + def wrapped(self, *args, **kwargs): + tensor = method(self, *args, **kwargs) + if isinstance(tensor, torch.Tensor): + return tensor.to(get_accelerator().current_device()) + return tensor + + return wrapped diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..45e523ab62b9b1b1a43c86685c5adfce59a718e6 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__init__.py @@ -0,0 +1,8 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .base_engine import CheckpointEngineBase +from .in_memory_engine import InMemoryModelEngine +from .huggingface_engine import HuggingFaceCheckpointEngine diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a95757cdc0764c68630ed5d599b1d8ae367e0a30 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/base_engine.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/base_engine.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..8f101c39ad87d99021246f390adcb85f868172c2 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/base_engine.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/huggingface_engine.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/huggingface_engine.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..cae851f8e05bfe98a18db68f20bf1cdf236825c1 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/huggingface_engine.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/in_memory_engine.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/in_memory_engine.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..3bb324c6262956e849e55e95d7dae8b2b7175d6d Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/__pycache__/in_memory_engine.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/base_engine.py b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/base_engine.py new file mode 100644 index 0000000000000000000000000000000000000000..26fc467d4d863aa0b6f6fb28b84d3f6260c702f2 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/base_engine.py @@ -0,0 +1,41 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from abc import ABC, abstractmethod +from typing import Iterable, Tuple + +import torch + +#from .huggingface_engine import HuggingFaceCheckpointEngine + +MEGATRON = 'megatron' +HUGGINGFACE = 'huggingface' + + +class CheckpointEngineBase(ABC): + """ + Abstract interface for checkpoint engines to implement. + + There is no ``__init__`` method here by design, since the creation of the checkpoint + engine will happen outside the policy/engine code. The tradeoff being made here is + that we will write different frontends for different checkpoint engines, but these + frontends can be tailored to the specific checkpoint engine/model source needs. + """ + + @abstractmethod + def parameters(self) -> Iterable[Tuple[str, torch.Tensor]]: + """ + This method should create a generator of tuples of the form (name, parameter) for + all parameters in the model. The name should be the fully qualified name of the + parameter, and the parameter should be a torch.Tensor. + + The expected use of a checkpoint engine is the following: + ```python + for name, parameter in checkpoint_engine.parameters(): + container_map.map_param(name, parameter) + ``` + For a concrete use example, see ``InferenceV2Policy``. + """ + ... diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/huggingface_engine.py b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/huggingface_engine.py new file mode 100644 index 0000000000000000000000000000000000000000..b17bb886838f1ed4efcfddd0f4f5f00f959ec2fa --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/huggingface_engine.py @@ -0,0 +1,130 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import os +import json +import torch +from .base_engine import CheckpointEngineBase +from typing import Iterable, Tuple +from functools import partial + +from ..logging import inference_logger + + +class HuggingFaceCheckpointEngine(CheckpointEngineBase): + + def __init__(self, model_name_or_path: str, auth_token: str = None, **hf_kwargs) -> None: + super().__init__() + from transformers import AutoConfig, GenerationConfig + + self.model_name_or_path = model_name_or_path + self.auth_token = auth_token + self.model_config = AutoConfig.from_pretrained(self.model_name_or_path, **hf_kwargs) + # Define this property here so we can use it in the model implementation + if not hasattr(self.model_config, "max_seq_length"): + if hasattr(self.model_config, "max_position_embeddings"): + self.model_config.max_seq_length = self.model_config.max_position_embeddings + else: + generation_config = GenerationConfig.from_pretrained(self.model_name_or_path) + self.model_config.max_seq_length = generation_config.max_length + self._local_checkpoint_dir = None + self._all_ckpt_paths = self._fetch_checkpoint_files() + + def _fetch_checkpoint_files(self): + """ + Fetch the checkpoint files from the HuggingFace Hub. + """ + # TODO(jeff): for models like llama-2 the user will have to provide an auth `token`, + # currently coming from the ckpt engine init but maybe a catch all kwargs for other + # snapshot download parameters would be more flexible. + + from huggingface_hub import snapshot_download, list_repo_tree + + def model_has_safetensors(model_name_or_path: str) -> bool: + if os.path.isdir(model_name_or_path): + file_list = os.listdir(model_name_or_path) + else: + file_list = [rf.path for rf in list_repo_tree(model_name_or_path)] + for f in file_list: + if f.endswith(".safetensors"): + return True + return False + + if os.path.isdir(self.model_name_or_path): + self._local_checkpoint_dir = self.model_name_or_path + else: + # We need to download the checkpoint files from HF + if model_has_safetensors(self.model_name_or_path): + # Prioritize downloading safetensors if they are available + allow_patterns = ["*.safetensors", "*.json"] + else: + # Fallback to bin files when safetensors are not present + allow_patterns = ["*.bin", "*.json", "*.pt"] + self._local_checkpoint_dir = snapshot_download(self.model_name_or_path, + allow_patterns=allow_patterns, + revision=None, + token=self.auth_token) + + assert os.path.isdir( + self._local_checkpoint_dir + ), f"Checkpoint dir {self._local_checkpoint_dir} is not a directory, cannot load checkpoint." + + # Set the appropriate file names based on whether we have safetensors or not + if model_has_safetensors(self._local_checkpoint_dir): + from safetensors.torch import load_file + model_param_json_fname = "model.safetensors.index.json" + model_file_fname = "model.safetensors" + self._checkpoint_load_fn = load_file + else: + model_param_json_fname = "pytorch_model.bin.index.json" + model_file_fname = "pytorch_model.bin" + self._checkpoint_load_fn = partial(torch.load, map_location="cpu", weights_only=False) + + model_param_json = os.path.join(self._local_checkpoint_dir, model_param_json_fname) + + if not os.path.isfile(model_param_json): + # We don't need any json as all such HF models will have pytorch_model.bin + all_checkpoint_files = [os.path.join(self._local_checkpoint_dir, model_file_fname)] + else: + param_map = json.load(open(model_param_json, "r")) + + # weight_map -> { "lm_head.weight": "pytorch_model-00002-of-00002.bin", ... } + weight_map = param_map["weight_map"] + + # unique set of all checkpoint files + all_checkpoint_files = set(weight_map.values()) + + # get absolute path of all unique checkpoint files + all_checkpoint_files = [os.path.join(self._local_checkpoint_dir, f) for f in all_checkpoint_files] + + return all_checkpoint_files + + def parameters(self) -> Iterable[Tuple[str, torch.Tensor]]: + """ + Generator of model parameters (satisfies the CheckpointEngineBase interface). + """ + for checkpoint in self._all_ckpt_paths: + inference_logger().info(f"Loading checkpoint: {checkpoint}") + checkpoint_sd = self._checkpoint_load_fn(checkpoint) + + # If the model has tied embeddings, we need to make sure the lm_head weights are tied to the embeddings weights + if hasattr(self.model_config, "tie_word_embeddings") and self.model_config.tie_word_embeddings: + if self.model_config.model_type == "qwen2": + checkpoint_sd["lm_head.weight"] = checkpoint_sd["model.embed_tokens.weight"] + + param_keys = list(checkpoint_sd.keys()) + for param_name in param_keys: + param = checkpoint_sd[param_name] + yield param_name, param + + del checkpoint_sd + + +if __name__ == "__main__": + # To test, add your auth_token here and run `python huggingface_engine.py` + engine = HuggingFaceCheckpointEngine(model_name_or_path="meta-llama/Llama-2-7b-hf", + auth_token="hf_xxxxxxxxxxxxxxxxx") + for name, param in engine.parameters(): + print(name, param.shape) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/in_memory_engine.py b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/in_memory_engine.py new file mode 100644 index 0000000000000000000000000000000000000000..13ec7b288f5febb158e167890e1f259d9684ed28 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/checkpoint/in_memory_engine.py @@ -0,0 +1,40 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Tuple +import torch + +from .base_engine import CheckpointEngineBase + + +class InMemoryModelEngine(CheckpointEngineBase): + """ + This "checkpoint" engine uses the existing interface to enable loading parameters into an + inference model from a model already instantiated in memory. In general, this is not the + recommended way to use the inference engine, and should only be used when absolutely necessary. + + The primary limitation of this approach is that the model must be fully instantiated in memory. + In a tensor parallel scenario, this means that the model is either replicated many times in host + memory. Currently, it is also recommended to only use this approach for models held in host memory. + + In order to free the memory held by this copy of the model, we delete the model in the first call + to `parameters`, so it is not safe to make this call twice. + """ + + def __init__(self, model: torch.nn.Module) -> None: + """ + Create virtual checkpoint engine for the provided module. + + Args: + model (torch.nn.Module): Model to load parameters from. + """ + super().__init__() + self.model = model + + def parameters(self) -> Iterable[Tuple[str, torch.Tensor]]: + for name, parameter in self.model.named_parameters(): + yield name, parameter + + del self.model diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/config_v2.py b/lib/python3.12/site-packages/deepspeed/inference/v2/config_v2.py new file mode 100644 index 0000000000000000000000000000000000000000..325b57d8f56a8730158096a2f8109a15b662ecf7 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/config_v2.py @@ -0,0 +1,44 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from pydantic import Field +from typing import Optional + +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from .ragged import DSStateManagerConfig + + +class DeepSpeedTPConfig(DeepSpeedConfigModel): + """ Configure tensor parallelism settings """ + + tp_size: int = 1 + """ Number of devices to split the model across using tensor parallelism. """ + + +class QuantizationConfig(DeepSpeedConfigModel): + """ Configure tensor parallelism settings """ + + quantization_mode: Optional[str] = None + """ The quantization mode in string format. The supported modes are as follows: + - 'wf6af16', weight-only quantization with FP6 weight and FP16 activation. + """ + # TODO: may reuse the constants in deepspeed/compression/constants.py + + +class RaggedInferenceEngineConfig(DeepSpeedConfigModel): + """ Sets parameters for DeepSpeed Inference Engine. """ + + tensor_parallel: DeepSpeedTPConfig = Field({}, alias="tp") + """ + Configuration for tensor parallelism used to split the model across several + GPUs. Expects a dictionary containing values for :any:`DeepSpeedTPConfig`. + """ + + state_manager: DSStateManagerConfig = Field({}, alias="manager") + """ + Configuration for managing persistent state + """ + + quantization: QuantizationConfig = {} diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/engine_factory.py b/lib/python3.12/site-packages/deepspeed/inference/v2/engine_factory.py new file mode 100644 index 0000000000000000000000000000000000000000..9c3188dfebb86e08aea555858b3873c14a9438aa --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/engine_factory.py @@ -0,0 +1,135 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import json +import logging +import os +import pickle +from packaging import version + +from .engine_v2 import InferenceEngineV2 +from .config_v2 import RaggedInferenceEngineConfig +from .checkpoint import HuggingFaceCheckpointEngine +from .logging import inference_logger +from .model_implementations import ( + OPTPolicy, + Llama2Policy, + MistralPolicy, + MixtralPolicy, + FalconPolicy, + PhiPolicy, + Phi3Policy, + QwenPolicy, + Qwen2Policy, + Qwen2MoePolicy, +) +from .model_implementations.inference_policy_base import POLICIES, InferenceV2Policy +from .model_implementations.flat_model_helpers import make_metadata_filename, ModelMetadata + + +def build_engine_from_ds_checkpoint(path: str, + engine_config: RaggedInferenceEngineConfig, + debug_level: int = logging.INFO) -> InferenceEngineV2: + """ + Creates an engine from a checkpoint saved by ``InferenceEngineV2``. + + Arguments: + path: Path to the checkpoint. This does not need to point to any files in particular, + just the directory containing the checkpoint. + engine_config: Engine configuration. See ``RaggedInferenceEngineConfig`` for details. + debug_level: Logging level to use. Unless you are actively seeing issues, the recommended + value is ``logging.INFO``. + + Returns: + Fully initialized inference engine ready to serve queries. + """ + + inference_logger(level=debug_level) + # Load metadata, for grabbing the policy name we'll have all ranks just check for + # rank 0. + metadata_filename = make_metadata_filename(path, 0, engine_config.tensor_parallel.tp_size) + metadata = json.load(open(metadata_filename, "r")) + metadata = ModelMetadata.parse_raw(metadata) + + # Get the policy + try: + policy_cls: InferenceV2Policy = POLICIES[metadata.policy] + except KeyError: + raise ValueError(f"Unknown policy {metadata.policy} for model {path}") + + # Load the model config + model_config = pickle.load(open(os.path.join(path, "ds_model_config.pkl"), "rb")) + policy = policy_cls(model_config, inf_checkpoint_path=path) + + return InferenceEngineV2(policy, engine_config) + + +def build_hf_engine(path: str, + engine_config: RaggedInferenceEngineConfig, + debug_level: int = logging.INFO) -> InferenceEngineV2: + """ + Build an InferenceV2 engine for HuggingFace models. This can accept both a HuggingFace + model name or a path to an Inference-V2 checkpoint. + + Arguments: + path: Path to the checkpoint. This does not need to point to any files in particular, + just the directory containing the checkpoint. + engine_config: Engine configuration. See ``RaggedInferenceEngineConfig`` for details. + debug_level: Logging level to use. Unless you are actively seeing issues, the recommended + value is ``logging.INFO``. + + Returns: + Fully initialized inference engine ready to serve queries. + """ + + if os.path.exists(os.path.join(path, "ds_model_config.pkl")): + return build_engine_from_ds_checkpoint(path, engine_config, debug_level=debug_level) + else: + # Set up logging + inference_logger(level=debug_level) + # get HF checkpoint engine + checkpoint_engine = HuggingFaceCheckpointEngine(path) + + # get model config from HF AutoConfig + model_config = checkpoint_engine.model_config + + # get the policy + # TODO: generalize this to other models + if model_config.model_type == "opt": + if not model_config.do_layer_norm_before: + raise ValueError( + "Detected OPT-350m model. This model is not currently supported. If this is not the 350m model, please open an issue: https://github.com/deepspeedai/DeepSpeed-MII/issues" + ) + policy = OPTPolicy(model_config, checkpoint_engine=checkpoint_engine) + elif model_config.model_type == "llama": + policy = Llama2Policy(model_config, checkpoint_engine=checkpoint_engine) + elif model_config.model_type == "mistral": + # Ensure we're using the correct version of transformers for mistral + import transformers + assert version.parse(transformers.__version__) >= version.parse("4.34.0"), \ + f"Mistral requires transformers >= 4.34.0, you have version {transformers.__version__}" + policy = MistralPolicy(model_config, checkpoint_engine=checkpoint_engine) + elif model_config.model_type == "mixtral": + # Ensure we're using the correct version of transformers for mistral + import transformers + assert version.parse(transformers.__version__) >= version.parse("4.36.1"), \ + f"Mistral requires transformers >= 4.36.1, you have version {transformers.__version__}" + policy = MixtralPolicy(model_config, checkpoint_engine=checkpoint_engine) + elif model_config.model_type == "falcon": + policy = FalconPolicy(model_config, checkpoint_engine=checkpoint_engine) + elif model_config.model_type == "phi": + policy = PhiPolicy(model_config, checkpoint_engine=checkpoint_engine) + elif model_config.model_type == "phi3": + policy = Phi3Policy(model_config, checkpoint_engine=checkpoint_engine) + elif model_config.model_type == "qwen": + policy = QwenPolicy(model_config, checkpoint_engine=checkpoint_engine) + elif model_config.model_type == "qwen2": + policy = Qwen2Policy(model_config, checkpoint_engine=checkpoint_engine) + elif model_config.model_type == "qwen2_moe": + policy = Qwen2MoePolicy(model_config, checkpoint_engine=checkpoint_engine) + else: + raise ValueError(f"Unsupported model type {model_config.model_type}") + + return InferenceEngineV2(policy, engine_config) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/engine_v2.py b/lib/python3.12/site-packages/deepspeed/inference/v2/engine_v2.py new file mode 100644 index 0000000000000000000000000000000000000000..4a358310377f00af00d92b6e5dcd2fd5d600392d --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/engine_v2.py @@ -0,0 +1,268 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import os +import json +import pickle +from typing import Iterable, Tuple + +import torch + +import deepspeed.comm as dist + +from deepspeed.accelerator import get_accelerator +from deepspeed.comm.comm import init_distributed + +from .model_implementations import InferenceV2Policy +from .logging import inference_logger +from .ragged import DSStateManager, RaggedBatchWrapper, PlaceholderSequenceDescriptor +from .scheduling_utils import SchedulingError, SchedulingResult +from .model_implementations.flat_model_helpers import make_param_filename, make_metadata_filename +from .model_implementations.inference_model_base import DSInferenceModelBase + +from .config_v2 import RaggedInferenceEngineConfig + +INFERENCE_MODEL_TIMER = "model-forward-inference" + + +class InferenceEngineV2: + + _config: RaggedInferenceEngineConfig + """ + Configuration of the inference engine. + """ + + _model: DSInferenceModelBase + """ + Inference model supporting ragged inference. + """ + + _state_manager: DSStateManager + """ + Persistent state manager for sequences and KV-cache. + """ + + @property + def free_blocks(self) -> torch.Tensor: + """ + Number of free KV blocks. This is a tensor of shape [n_kv_cache_groups] where each + element is the number of free blocks in the corresponding KV cache group. + """ + return self._state_manager.free_blocks + + @property + def n_kv_cache_groups(self) -> int: + """ + Number of KV cache groups. + """ + return self._state_manager.n_kv_cache_groups + + def model(self) -> DSInferenceModelBase: + """ + The model implementation. + """ + return self._model + + def __init__(self, policy: InferenceV2Policy, engine_config: RaggedInferenceEngineConfig) -> None: + """ + Create the Inference V2 engine. + + Arguments: + policy (InferenceV2Policy): Policy for the model implementation. This policy object + will be used to build the model and load the checkpoint associated with it. + engine_config (RaggedInferenceEngineConfig): Configuration for the inference engine. + """ + self._config = engine_config + self._policy = policy + self._base_mp_group = self._initialize_tp_group() + + # Build model from policy + inference_logger().info("Building model...") + self._model = self._policy.build_model(self._config, self._base_mp_group) + inference_logger().info("Model built.") + + # Create state manager + self._batch = RaggedBatchWrapper(self._config.state_manager) + self._state_manager = DSStateManager(self._config.state_manager, + self._model.kv_cache_config(), + base_mp_group=self._base_mp_group) + self._model.set_state_manager(self._state_manager) + + def _initialize_tp_group(self): + """ + Implementation of our TP group initialization. + """ + init_distributed() + local_rank = int(os.getenv("LOCAL_RANK", 0)) + get_accelerator().set_device(local_rank) + + if local_rank >= self._config.tensor_parallel.tp_size: + raise RuntimeError("Local rank is greater than TP size, ensure that the TP config is correct.") + + ranks = list(range(self._config.tensor_parallel.tp_size)) + return dist.new_group(ranks=ranks) + + def put(self, + batch_uids: Iterable[int], + batch_tokens: Iterable[torch.Tensor], + do_checks: bool = True) -> torch.Tensor: + """ + Put a ragged batch onto the inference engine. This will perform one forward and return + a Tensor of the shape [len(batch_uids), *output_shape]. Logits for the non-final tokens + are not calculated. + + Arguments: + batch_uids: Iterable of uids for the batch on the host + batch_tokens: Iterable of token tensors for the batch on the host + do_checks: Check schedulability when it is set to True. You can skip this check for better performance when it has already been completed. + """ + + if do_checks: + token_lens = [len(tokens) for tokens in batch_tokens] + schedule_check = self.can_schedule(batch_uids, token_lens) + if schedule_check != SchedulingResult.Success: + raise SchedulingError(schedule_check) + + self._batch.clear() + for uid, tokens in zip(batch_uids, batch_tokens): + + host_seq_desc = self._state_manager.get_or_create_sequence(uid) + self._model.maybe_allocate_kv(host_seq_desc, tokens.numel()) + host_seq_desc.pre_forward(tokens.numel()) + + # We can disable checks since we already validated schedulability. + self._batch.insert_sequence(host_seq_desc, tokens, do_checks=do_checks) + + # Send all metadata to the device + self._batch.finalize() + + # Prep all data structures for the actual forward (in anticipation of CG in the future) + # and also to amortize some of the costs in a more straightforward way. + self._model.prepare_batch(self._batch) + + # Model implementation will pick up in the forward. + logits = self._model.forward(self._batch) + + # We return one set of logits per sequence in the batch (saves cost on unembedding) + assert logits.shape[0] == self._batch.current_sequences + + for uid in batch_uids: + host_seq_desc = self._state_manager.get_sequence(uid) + host_seq_desc.post_forward() # Updates sequence metadata. + self._model.maybe_free_kv(host_seq_desc) + + return logits + + def query(self, uid: int, max_request_tokens: int, max_request_blocks) -> Tuple[int, torch.Tensor]: + """ + Determine the number of tokens and KV blocks to reserve for a given request. Given a UID + (this UID may not be recognized by the model yet), this will return the number of tokens + and blocks to reserve for the request. + + Arguments: + uid (int): The UID of the sequence (as tracked by the scheduling entity). If + this is a new sequence (with a UID unknown to the inference engine), then + an empty placeholder is created to pass to the occupancy logic. + n_tokens (int): The number of tokens to hypothetically send. + + Returns: + Tuple[int, Optional[int]]: Tuple of free kv blocks and the number of blocks + required to schedule the sequence. + """ + seq_desc = self._state_manager.get_sequence(uid) + if seq_desc is None: + if (self._state_manager.n_tracked_sequences == self._config.state_manager.max_tracked_sequences): + return (0, 0) + seq_desc = PlaceholderSequenceDescriptor() + + req_tokens, req_blocks = self._model.get_kv_requirements(seq_desc, max_request_tokens, max_request_blocks) + + return (req_tokens, req_blocks) + + def can_schedule(self, uids: Iterable[int], lengths: Iterable[int]) -> SchedulingResult: + """ + Dry run a batch to determine if it can be scheduled. Placeholder sequences will be + created for any UIDs that are unknown to the inference engine. + + Arguments: + uids (Iterable[int]): Iterable of UIDs for the batch + lengths (Iterable[int]): Iterable of lengths for each sequence of the batch. This lengths + corresponds to the number of tokens to send in the hypothetical forward; history + tokens will be determined via UID lookup and future tokens are disregarded. + + Returns: + bool: True if the batch can be scheduled, False otherwise. + """ + + cur_seqs = self._state_manager.n_tracked_sequences + free_blocks = self._state_manager.free_blocks + req_blocks = 0 + batch_len = 0 + + if len(uids) > self._config.state_manager.max_ragged_sequence_count: + # Can only compose a batch from a limited number of sequences + return SchedulingResult.BatchSequenceLimitExceeded + + for uid, length in zip(uids, lengths): + seq_desc = self._state_manager.get_sequence(uid) + if seq_desc is None: + cur_seqs += 1 + seq_desc = PlaceholderSequenceDescriptor() + + sched_len, sched_blocks = self._model.get_kv_requirements(seq_desc, length, free_blocks) + + if sched_len != length: + # We ran out of KV cache + return SchedulingResult.KVCacheLimitExceeded + + batch_len += length + free_blocks -= sched_blocks + + if cur_seqs > self._config.state_manager.max_tracked_sequences: + # Would run out of tracking metadata + return SchedulingResult.EngineSequenceLimitExceeded + + if batch_len > self._config.state_manager.max_ragged_batch_size: + # Would exceed the maximum batch size + return SchedulingResult.BatchTokenLimitExceeded + + return SchedulingResult.Success + + def get_remaining_block_capacity(self, uid: int) -> int: + """ + Get the remaining capacity of the last block already allocated. + """ + seq_desc = self._state_manager.get_sequence(uid) + if seq_desc is None: + return 0 + return self._model.get_remaining_block_capacity(seq_desc) + + def flush(self, uid: int) -> None: + """ + Remove all state associated with a sequence from the inference engine. + + Arguments: + uid (int): The UID of the sequence to flush. + """ + self._state_manager.flush_sequence(uid) + + def serialize(self, save_path: str) -> None: + """ + Serialize the model to a file. + + Arguments: + path (str): Path to the file to serialize to. + """ + param_file_name = make_param_filename(save_path, self._model.tp_rank, self._model.tp_size) + metadata_file_name = make_metadata_filename(save_path, self._model.tp_rank, self._model.tp_size) + + # Save the flattened parameters + + torch.save(self._model.flattened_params, param_file_name) + + json.dump(self._model.flattened_param_metadata.json(), open(metadata_file_name, "w")) + + if self._model.tp_rank == 0: + pickle.dump(self._model._config, open(os.path.join(save_path, "ds_model_config.pkl"), "wb")) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/inference_parameter.py b/lib/python3.12/site-packages/deepspeed/inference/v2/inference_parameter.py new file mode 100644 index 0000000000000000000000000000000000000000..4dcff16a4515ce37ed334c9b0b0d623eea5b2ac2 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/inference_parameter.py @@ -0,0 +1,89 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Dict + +import torch + +CORE_PARAM = "_ds_core_param_key" + +STR_TO_DTYPE = { + "torch.float32": torch.float32, + "torch.float64": torch.float64, + "torch.float16": torch.float16, + "torch.bfloat16": torch.bfloat16, + "torch.int64": torch.int64, + "torch.int32": torch.int32, + "torch.int16": torch.int16, + "torch.int8": torch.int8, + "torch.uint8": torch.uint8, + "torch.bool": torch.bool, +} + + +class InferenceParameter(torch.Tensor): + """ + An extension of the torch.Tensor class to support our inference focused features. One important + thing to note here is that an InferenceParam can be used a torch.Tensor, but outputs of + torch.Tensor operations will not be InferenceParams. + """ + + @staticmethod + def __new__(cls, tensor, *args, **kwargs): + new_tensor = super().__new__(cls, tensor, *args, **kwargs) + if hasattr(tensor, "_aux_attrs"): + setattr(new_tensor, "_aux_attrs", tensor.aux_attrs) + return new_tensor + + def to(self, *args, **kwargs): + new_tensor = super().to(*args, **kwargs) + if hasattr(self, "_aux_attrs"): + setattr(new_tensor, "_aux_attrs", self.aux_attrs) + try: + _ = torch.device(args[0]) + for name, attr in new_tensor.aux_attrs.items(): + new_attr = attr.to(*args, **kwargs) + setattr(new_tensor, name, new_attr) + new_tensor.aux_attrs[name] = new_attr + except: + pass + + return new_tensor + + @classmethod + def initialize(cls, core_param: torch.Tensor, **kwargs) -> 'InferenceParameter': + """ + Create the inference parameter. + """ + param = InferenceParameter(core_param) + setattr(param, "_aux_attrs", kwargs) + + for attr_name, attr in kwargs.items(): + if hasattr(param, attr_name): + raise ValueError(f"Attribute {attr_name} already exists on param.") + + if not isinstance(attr, torch.Tensor): + raise ValueError(f"Attribute {attr_name} must be a tensor.") + + setattr(param, attr_name, attr) + + return param + + @classmethod + def initialize_raw(self, **kwargs) -> 'InferenceParameter': + """ + All kwargs must be torch.Tensors and must include the core parameter. + """ + if CORE_PARAM not in kwargs: + raise ValueError(f"Must provide core parameter, with key {CORE_PARAM}.") + + return InferenceParameter.initialize(kwargs[CORE_PARAM], **kwargs) + + @property + def aux_attrs(self) -> Dict[str, torch.Tensor]: + """ + Dictionary of auxiliary attributes. + """ + return self._aux_attrs diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/inference_utils.py b/lib/python3.12/site-packages/deepspeed/inference/v2/inference_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7b2dd4237353d85b4249faafb2d8c3051289b2ef --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/inference_utils.py @@ -0,0 +1,105 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Dict + +import torch + +from enum import Enum, IntEnum + + +class NormTypeEnum(Enum): + LayerNorm: str = "layer_norm" + RMSNorm: str = "rms_norm" + + +class DtypeEnum(Enum): + # The torch dtype must always be the first value (so we return torch.dtype) + fp16 = torch.float16, "torch.float16", "fp16", "float16", "half" + fp32 = torch.float32, "torch.float32", "fp32", "float32", "float" + bf16 = torch.bfloat16, "torch.bfloat16", "bf16", "bfloat16", "bfloat" + int8 = torch.int8, "torch.int8", "int8" + + # Copied from https://stackoverflow.com/a/43210118 + # Allows us to use multiple values for each Enum index and returns first + # listed value when Enum is called + def __new__(cls, *values): + obj = object.__new__(cls) + # first value is canonical value + obj._value_ = values[0] + for other_value in values[1:]: + cls._value2member_map_[other_value] = obj + obj._all_values = values + return obj + + def __repr__(self): + return "<%s.%s: %s>" % ( + self.__class__.__name__, + self._name_, + ", ".join([repr(v) for v in self._all_values]), + ) + + +ELEM_SIZES: Dict[torch.dtype, int] = { + torch.float16: 2, + torch.bfloat16: 2, + torch.float32: 4, + torch.float64: 8, + torch.int8: 1, + torch.uint8: 1, + torch.int16: 2, + torch.int32: 4, + torch.int64: 8, + torch.bool: 1, +} + + +class ActivationType(IntEnum): + """ + Types of activations supported by DS-Inference + """ + + GELU = 0 + + RELU = 1 + + SILU = 2 + + GEGLU = 3 + + ReGLU = 4 + + SiGLU = 5 + + IDENTITY = 6 + + InvalidType = -1 + + +def is_gated(act_fn: ActivationType) -> bool: + """ + Return True if the given activation function is gated. + """ + if not isinstance(act_fn, ActivationType): + act_fn = ActivationType(act_fn) + + return act_fn in [ActivationType.GEGLU, ActivationType.ReGLU, ActivationType.SiGLU] + + +def elem_size(dtype: torch.dtype) -> int: + """ + Return size in bytes of the given dtype. + """ + try: + return ELEM_SIZES[dtype] + except KeyError: + raise ValueError("Unknown dtype size for {}".format(dtype)) + + +def ceil_div(a: int, b: int) -> int: + """ + Return ceil(a / b). + """ + return -(-a // b) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/logging.py b/lib/python3.12/site-packages/deepspeed/inference/v2/logging.py new file mode 100644 index 0000000000000000000000000000000000000000..77afe351cbea127c2f0f5dbc9a249dd71516ca31 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/logging.py @@ -0,0 +1,26 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import logging + +from deepspeed.utils.logging import LoggerFactory + +inf_logger = None + + +def inference_logger(level: int = logging.INFO) -> logging.Logger: + """ + Create the inference logger. NOTE: Logging is not cost free. On a 3960X, + there is a cost of about 6 us per call to a no-op logger, so this should + be used during setup only and not during the inference loop. + + Args: + level (int, optional): The logging level. Defaults to logging.INFO. + """ + global inf_logger + if inf_logger is None: + inf_logger = LoggerFactory.create_logger(name="DS-Inference", level=level) + inf_logger.debug("Inference logger created.") + return inf_logger diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..3483d9348c55599f8066be65b5d89ef04aade950 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__init__.py @@ -0,0 +1,21 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .inference_model_base import DSInferenceModelBase +from .inference_transformer_base import DSTransformerModelBase, DSMoETransformerModelBase +from .inference_policy_base import InferenceV2Policy, ContainerMap +from .sharding import * + +# Model Implementations +from .llama_v2 import * +from .opt import * +from .mistral import * +from .mixtral import * +from .falcon import * +from .phi import * +from .phi3 import * +from .qwen import * +from .qwen_v2 import * +from .qwen_v2_moe import * diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d3cada8ec879b5eb7c610d0ec909496617c36626 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/flat_model_helpers.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/flat_model_helpers.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c53e4951d294bd14d43520855b5787abfbde3ef5 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/flat_model_helpers.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/inference_model_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/inference_model_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..9a333ff37634f6cbba96f8e7903a9eb1d1947f65 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/inference_model_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/inference_policy_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/inference_policy_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ebf89d595b83cf95b416fa7c12b207ad97298f0a Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/inference_policy_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/inference_transformer_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/inference_transformer_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..0c0089279fd8db85179b4b7c5dea8732e2171c1f Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/inference_transformer_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/layer_container_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/layer_container_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d101a739eca8911db9ed5c7bdd4645c08efa594d Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/layer_container_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/parameter_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/parameter_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b40cbc9f3ff2696dd943ebdc6812cbc9813b1bb8 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/__pycache__/parameter_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..60963011cd660fe5e43b9a90efdacec4b16651b9 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .attn_output_parameters import * +from .embedding_parameters import * +from .mlp_parameters import * +from .moe_parameters import * +from .norm_parameters import * +from .qkv_parameters import * +from .unembed_parameters import * +from .invfreq_parameters import * diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..48c3b41e174228aab9c5c11f9bd54f99e87fa864 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/attn_output_parameters.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/attn_output_parameters.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..85f45842c1718f566be9deea4de4dbda54f08115 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/attn_output_parameters.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/embedding_parameters.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/embedding_parameters.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..da147e1ea15ce597eda4b2aaee5e71f2f46c0356 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/embedding_parameters.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/invfreq_parameters.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/invfreq_parameters.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d859adb6ff5217251b7f4f201cf225eb7728e5f3 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/invfreq_parameters.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/mlp_parameters.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/mlp_parameters.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f4c35ac9fa3db343151c8e60d53812b31277aced Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/mlp_parameters.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/moe_parameters.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/moe_parameters.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..cb735e6f85946da7720aaacab5e2566ae0a73f4b Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/moe_parameters.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/norm_parameters.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/norm_parameters.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..365f77e7e232d9bd312544c160d261f3da0b798e Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/norm_parameters.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/qkv_parameters.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/qkv_parameters.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d7ac24786321bfac1832d5d1767a4229750821d1 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/qkv_parameters.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/unembed_parameters.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/unembed_parameters.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..8c371990eef35badc56137039759ec137c04ad66 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/__pycache__/unembed_parameters.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/attn_output_parameters.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/attn_output_parameters.py new file mode 100644 index 0000000000000000000000000000000000000000..f220cf7a7125d030b05829d0615210c79e9d562d --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/attn_output_parameters.py @@ -0,0 +1,29 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from ...model_implementations.parameter_base import ParameterBase +""" +Common Attention Output Parameter Patterns +""" + + +class AttentionOutputParameter(ParameterBase): + """ + Attention output parameter container. + + Note: The differentiation for something like GQA for this matrix is primarily + encompassed in the sharding logic, which is currently expected to be performed by + the model implementation. + """ + + params: torch.Tensor + """ + Unsharded attention output parameter of shape [model_dim, model_dim] + """ + + def finalize(self) -> torch.Tensor: + return self.inference_model.transform_attn_out_param(self.params) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/embedding_parameters.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/embedding_parameters.py new file mode 100644 index 0000000000000000000000000000000000000000..2ed34b5fd259a77e858073a32deef1f805dd1325 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/embedding_parameters.py @@ -0,0 +1,26 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from ...model_implementations.parameter_base import ParameterBase +""" +Embedding containers. +""" + + +class EmbeddingParameter(ParameterBase): + """ + Embedding container. This should be safe to use for all types of embeddings (i.e. word, position, + and token type). + """ + + params: torch.Tensor + """ + Vocabulary parameter of shape [vocab_size, model_dim]. + """ + + def finalize(self) -> torch.Tensor: + return self.inference_model.transform_embedding_param(self.params) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/invfreq_parameters.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/invfreq_parameters.py new file mode 100644 index 0000000000000000000000000000000000000000..163f9de81d98bc30e8fb7bbf3b0b724dd74cd0c3 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/invfreq_parameters.py @@ -0,0 +1,19 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from ...model_implementations.parameter_base import ParameterBase +""" +Common InvFreq Parameter Patterns +""" + + +class InvFreqParameter(ParameterBase): + + params: torch.Tensor + + def finalize(self) -> torch.Tensor: + return self.params.to(self.inference_model.activation_dtype.value) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/mlp_parameters.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/mlp_parameters.py new file mode 100644 index 0000000000000000000000000000000000000000..17def1fa021fb1ce8c23b9c30ca5c27c40d67488 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/mlp_parameters.py @@ -0,0 +1,99 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from ...model_implementations.parameter_base import ParameterBase +""" +MLP Parameter Containers +""" + + +class MLP1Parameter(ParameterBase): + """ + First MLP projection weight container. This performs a straight pass-through to the + model implementation for transformation. + """ + params: torch.Tensor + + def finalize(self) -> torch.Tensor: + # NOTE(cmikeh2): If we are gated but not in the format specified below, we should trigger a permutation here. + # I am not currently aware of any models that use this format (or how we should even detect it; probably should + # just be a different param entirely, but until then we'll just assume the format is correct). + return self.inference_model.transform_mlp_1_param(self.params) + + +class GatedMLPParameter(ParameterBase): + """ + Gated MLP projection container. + """ + + gate_params: torch.Tensor + """ + Weight parameter for the gating matrix. + """ + + up_params: torch.Tensor + """ + For lack of a better name, the non-gating weight parameters. + """ + + def finalize(self) -> torch.Tensor: + """ + Our gated format (this is different from InferenceV1!) is to have the gate and activated neurons + interleaved. So if we have 4 output neurons (two effective neurons) with 4 input neurons, the finalized + parameter will look like: + [g0_0, g0_1, g0_2, g0_3] + [a0_0, a0_1, a0_2, a0_3] + [g1_0, g1_1, g1_2, g1_3] + [a1_0, a1_1, a1_2, a1_3] + + As a reference, in inference v1, the format is: + [g0_0, g0_1, g0_2, g0_3] + [g1_0, g1_1, g1_2, g1_3] + [a0_0, a0_1, a0_2, a0_3] + [a1_0, a1_1, a1_2, a1_3] + """ + assert self.gate_params.shape[0] == self.up_params.shape[ + 0], "Gated MLP parameters must have the same number of neurons." + total_neurons = self.gate_params.shape[0] + self.up_params.shape[0] + + # flip the order if even with the correct tokenizer we get wrong output + #fused_param = torch.cat([self.up_params, self.gate_params], dim=-1).reshape(total_neurons, -1) + fused_param = torch.cat([self.gate_params, self.up_params], dim=-1).reshape(total_neurons, -1) + return self.inference_model.transform_mlp_1_param(fused_param) + + +class FusedGatedMLPParameter(ParameterBase): + """ + Gated MLP projection container. + """ + + params: torch.Tensor + """ + Weight parameter for the fused gating and non-gating weight parameters. + """ + + def finalize(self) -> torch.Tensor: + gate_params = self.params[:self.params.shape[0] // 2] + up_params = self.params[self.params.shape[0] // 2:] + total_neurons = gate_params.shape[0] + up_params.shape[0] + fused_param = torch.cat([gate_params, up_params], dim=-1).reshape(total_neurons, -1) + return self.inference_model.transform_mlp_1_param(fused_param) + + +class MLP2Parameter(ParameterBase): + """ + Second MLP projection weight container. This performs a straight pass-through to the + model implementation for transformation. + """ + + params: torch.Tensor + """ + Full weight parameter. + """ + + def finalize(self) -> torch.Tensor: + return self.inference_model.transform_mlp_2_param(self.params) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/moe_parameters.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/moe_parameters.py new file mode 100644 index 0000000000000000000000000000000000000000..8ababf567ba9a499624c5924dd32564ad82922fb --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/moe_parameters.py @@ -0,0 +1,78 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from ...model_implementations.parameter_base import ParameterBase, ParamList +""" +Moe Parameters + +These parameters are compatible with any model inheriting from ``DSMoETransformerModelBase``. +""" + + +class MoEGatingWeightParameter(ParameterBase): + """ + Gating weight matrix. + """ + + params: torch.Tensor + """ + Projection matrix from the input activations to the gate logits. + """ + + def finalize(self) -> torch.Tensor: + return self.inference_model.transform_moe_gate_param(self.params) + + +class UnfusedMoEMLP1Parameter(ParameterBase): + """ + This container should be used when the experts are held in separate parameters + and need to be joined into a single group. + """ + + experts: ParamList("n_experts") # noqa: F821 + + def finalize(self) -> torch.Tensor: + stacked_experts = torch.stack([p for p in self.experts], dim=0) + return self.inference_model.transform_moe_mlp_1_param(stacked_experts) + + +class UnfusedMoEMLP2Parameter(ParameterBase): + """ + This container should be used when the experts are held in separate parameters + and need to be joined into a single group. + """ + + experts: ParamList("n_experts") # noqa: F821 + + def finalize(self) -> torch.Tensor: + stacked_experts = torch.stack([p for p in self.experts], dim=0) + return self.inference_model.transform_moe_mlp_2_param(stacked_experts) + + +class UnfusedMoEGatedMLPParameter(ParameterBase): + """ + MoE Parameter for a gated activation function in which the gating matrix is not + fused in the same parameter as the non-gating matrix. + + This is a stacked version of the ``GatedMLPParameter``. Please see that class for more + documentation on the layout of the parameters. + """ + + gating_experts: ParamList("n_experts") # noqa: F821 + + up_experts: ParamList("n_experts") # noqa: F821 + + def finalize(self) -> torch.Tensor: + transposed_experts = [] + for gate, up in zip(self.gating_experts, self.up_experts): + assert gate.shape[0] == up.shape[0], "Gated MLP parameters must have the same number of neurons." + total_neurons = gate.shape[0] + up.shape[0] + fused_expert = torch.cat([gate, up], dim=-1).reshape(total_neurons, -1) + transposed_experts.append(fused_expert) + + stacked_experts = torch.stack(transposed_experts, dim=0) + return self.inference_model.transform_moe_mlp_1_param(stacked_experts) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/norm_parameters.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/norm_parameters.py new file mode 100644 index 0000000000000000000000000000000000000000..81ffcc3221df2dd2c05f9d8739a905a5ea2399a5 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/norm_parameters.py @@ -0,0 +1,22 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from ...model_implementations.parameter_base import ParameterBase +""" +Common Attention Output Parameter Patterns +""" + + +class NormParameter(ParameterBase): + """ + Simple normalization container. + """ + + params: torch.Tensor + + def finalize(self) -> torch.Tensor: + return self.inference_model.transform_norm_param(self.params) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/qkv_parameters.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/qkv_parameters.py new file mode 100644 index 0000000000000000000000000000000000000000..e240137186fe6114cb58ea66c1a25da59184d8f7 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/qkv_parameters.py @@ -0,0 +1,115 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from ...model_implementations.parameter_base import ParameterBase +""" +Common QKV Parameter Patterns +""" + + +class FusedQKVParameter(ParameterBase): + """ + Traditional fused QKV parameters for QKV projection. This is functionally + a direct copy. + + src_qkv_w shape: [3 * out_features, in_features] + qkv_w shape: [3 * out_features, in_features] + """ + + params: torch.Tensor + + def finalize(self) -> torch.Tensor: + return self.inference_model.transform_qkv_param(self.params) + + +class UnfusedQKVParameter(ParameterBase): + """ + QKV parameter container for unfused QKV projection. + + src_param shapes: 3 x [out_features, in_features] + dst_param shape: [3 x out_features, in_features] + """ + + q_params: torch.Tensor + + k_params: torch.Tensor + + v_params: torch.Tensor + + def finalize(self): + fused_param = torch.cat([self.q_params, self.k_params, self.v_params], dim=0) + return self.inference_model.transform_qkv_param(fused_param) + + +def megatron_qkv_reshape(param: torch.Tensor, head_size: int, n_heads: int) -> torch.Tensor: + assert param.shape[0] == 3 * n_heads * head_size + + all_heads = torch.chunk(param, chunks=3 * n_heads, dim=0) + q_heads = all_heads[::3] + k_heads = all_heads[1::3] + v_heads = all_heads[2::3] + return torch.cat([q_heads, k_heads, v_heads], dim=0) + + +class MegatronQKVParameter(ParameterBase): + """ + QKV parameter container for Megatron-style QKV projection. Megatron stores the parameter + as [n_heads, 3, head_size, in_features] whereas our inference system is built around + [3, n_heads, head_size, in_features]. This container handles the conversion. + + Note: this container expects the model implementation to implement properties for + `head_size` and `n_heads`. + + src_qkv_w shape: [3 * out_features, in_features] + qkv_w shape: [3 * out_features, in_features] + """ + + params: torch.Tensor + + def finalize(self) -> torch.Tensor: + head_size = self.inference_model.head_size + n_heads = self.inference_model.n_heads + + transposed_param = megatron_qkv_reshape(self.params, head_size, n_heads) + return self.inference_model.transform_qkv_param(transposed_param) + + +def transform_gqa_megatron(src_param: torch.Tensor, head_size: int, n_q_heads: int, n_kv_heads: int) -> torch.Tensor: + assert src_param.shape[0] == (2 * n_kv_heads + n_q_heads) * head_size + + head_ratio = n_q_heads // n_kv_heads + + # Reshape to get the groups as the leading dimension + groups_leading_view = src_param.reshape(n_kv_heads, 2 + head_ratio, head_size, -1) + q_heads = groups_leading_view[:, :head_ratio, :, :].reshape(-1, groups_leading_view.shape[-1]) + k_heads = groups_leading_view[:, head_ratio, :, :].reshape(-1, groups_leading_view.shape[-1]) + v_heads = groups_leading_view[:, head_ratio + 1, :, :].reshape(-1, groups_leading_view.shape[-1]) + # Squeeze will remove extra dimension for bias + return torch.cat([q_heads, k_heads, v_heads], dim=0).squeeze() + + +class GQAMegatronQKVParameter(ParameterBase): + """ + QKV parameter for Megatron-style QKV projection with GQA-style QKV projection. In this + storage format each of the groups is stored consecutively, so there will be multiple q_heads, + then one k head, and one v head. + + Note: this container expects the model implementation to implement properties for + `head_size`, `n_q_heads`, and `n_kv_heads`. + + src_qkv_w shape: [(2 * n_kv_heads + n_q_heads) * head_size, in_features] + qkv_w shape: [(2 * n_kv_heads + n_q_heads) * head_size, in_features] + """ + + params: torch.Tensor + + def finalize(self) -> torch.Tensor: + head_size = self.inference_model.head_size + n_q_heads = self.inference_model.n_heads_q + n_kv_heads = self.inference_model.n_heads_kv + transposed_param = transform_gqa_megatron(self.params, head_size, n_q_heads, n_kv_heads) + return self.inference_model.transform_qkv_param(transposed_param) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/unembed_parameters.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/unembed_parameters.py new file mode 100644 index 0000000000000000000000000000000000000000..9f67c0ce3c27d2fc07fc735acba8a64bda8bcbf1 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/common_parameters/unembed_parameters.py @@ -0,0 +1,26 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from ...model_implementations.parameter_base import ParameterBase +""" +Unembedding containers. +""" + + +class UnembedParameter(ParameterBase): + """ + Unembedding parameter. This will likely be mapped to the same original weight in the model as the + embedding, but we have a different preferred sharding approach. + """ + + params: torch.Tensor + """ + Unembedding parameter of shape [vocab_size, model_dim]. + """ + + def finalize(self) -> torch.Tensor: + return self.inference_model.transform_unembed_param(self.params) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..20f37538274ccda7ab68ceb7e82c87675120b7e7 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .policy import FalconPolicy diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..5dab82d4a4c78030d2fdcc31661680f9e874324e Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/container.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/container.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..8b581cb683cc7506bb68e11508e278283938d963 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/container.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/model.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/model.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..363239732dbb4c4d66a683b77659eb49f20b1f9e Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/model.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/policy.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/policy.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b98950e6c7b451a4dbc7e873ea96d35d16877c62 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/__pycache__/policy.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/container.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/container.py new file mode 100644 index 0000000000000000000000000000000000000000..caccfe1ecb00c37b08121decb250b2625c64eb9a --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/container.py @@ -0,0 +1,129 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# Create a container object to save model-specific tensors using the policy file above. + +from ..common_parameters import * +from ..layer_container_base import LayerContainer +''' + # HF Falcon 7b model looks like this: + +FalconForCausalLM( + (transformer): FalconModel( + (word_embeddings): Embedding(65024, 4544) + (h): ModuleList( + (0-31): 32 x FalconDecoderLayer( + (self_attention): FalconAttention( + (maybe_rotary): FalconRotaryEmbedding() + (query_key_value): FalconLinear(in_features=4544, out_features=4672, bias=False) + (dense): FalconLinear(in_features=4544, out_features=4544, bias=False) + (attention_dropout): Dropout(p=0.0, inplace=False) + ) + (mlp): FalconMLP( + (dense_h_to_4h): FalconLinear(in_features=4544, out_features=18176, bias=False) + (act): GELU(approximate='none') + (dense_4h_to_h): FalconLinear(in_features=18176, out_features=4544, bias=False) + ) + (input_layernorm): LayerNorm((4544,), eps=1e-05, elementwise_affine=True) + ) + ) + (ln_f): LayerNorm((4544,), eps=1e-05, elementwise_affine=True) + ) + (lm_head): Linear(in_features=4544, out_features=65024, bias=False) +) +''' + + +class FalconTransformerContainer(LayerContainer): + """ + Transformer layer container for the Falcon model. + """ + qkv_w: FusedQKVParameter + attn_out_w: AttentionOutputParameter + mlp_1_w: MLP1Parameter + mlp_2_w: MLP2Parameter + ln_attn_gamma: NormParameter + ln_attn_beta: NormParameter + + PARAM_MAPPING = { + "self_attention.query_key_value.weight": "qkv_w.params", + "self_attention.dense.weight": "attn_out_w.params", + "mlp.dense_h_to_4h.weight": "mlp_1_w.params", + "mlp.dense_4h_to_h.weight": "mlp_2_w.params", + "input_layernorm.weight": "ln_attn_gamma.params", + "input_layernorm.bias": "ln_attn_beta.params", + } + + +class FalconNonTransformerContainer(LayerContainer): + """ + Non-Transformer layer container for the Falcon model. + """ + word_emb: EmbeddingParameter + word_unembed: UnembedParameter + final_norm_gamma: NormParameter + final_norm_beta: NormParameter + + PARAM_MAPPING = { + "transformer.word_embeddings.weight": "word_emb.params", + "transformer.ln_f.weight": "final_norm_gamma.params", + "transformer.ln_f.bias": "final_norm_beta.params", + "lm_head.weight": "word_unembed.params", + } + + +''' + # HF Falcon 40b model looks like this: + + FalconForCausalLM( + (transformer): FalconModel( + (word_embeddings): Embedding(65024, 8192) + (h): ModuleList( + (0-59): 60 x FalconDecoderLayer( + (self_attention): FalconAttention( + (maybe_rotary): FalconRotaryEmbedding() + (query_key_value): FalconLinear(in_features=8192, out_features=9216, bias=False) + (dense): FalconLinear(in_features=8192, out_features=8192, bias=False) + (attention_dropout): Dropout(p=0.0, inplace=False) + ) + (mlp): FalconMLP( + (dense_h_to_4h): FalconLinear(in_features=8192, out_features=32768, bias=False) + (act): GELU(approximate='none') + (dense_4h_to_h): FalconLinear(in_features=32768, out_features=8192, bias=False) + ) + (ln_attn): LayerNorm((8192,), eps=1e-05, elementwise_affine=True) + (ln_mlp): LayerNorm((8192,), eps=1e-05, elementwise_affine=True) + ) + ) + (ln_f): LayerNorm((8192,), eps=1e-05, elementwise_affine=True) + ) + (lm_head): Linear(in_features=8192, out_features=65024, bias=False) +) +''' + + +class FalconNewArchTransformerContainer(LayerContainer): + """ + Transformer layer container for the Falcon model. + """ + qkv_w: GQAMegatronQKVParameter + attn_out_w: AttentionOutputParameter + mlp_1_w: MLP1Parameter + mlp_2_w: MLP2Parameter + ln_attn_gamma: NormParameter + ln_attn_beta: NormParameter + ln_mlp_gamma: NormParameter + ln_mlp_beta: NormParameter + + PARAM_MAPPING = { + "self_attention.query_key_value.weight": "qkv_w.params", + "self_attention.dense.weight": "attn_out_w.params", + "mlp.dense_h_to_4h.weight": "mlp_1_w.params", + "mlp.dense_4h_to_h.weight": "mlp_2_w.params", + "ln_attn.weight": "ln_attn_gamma.params", + "ln_attn.bias": "ln_attn_beta.params", + "ln_mlp.weight": "ln_mlp_gamma.params", + "ln_mlp.bias": "ln_mlp_beta.params", + } diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/model.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/model.py new file mode 100644 index 0000000000000000000000000000000000000000..b2830c80b562546d138430da702b5c9882ab00b6 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/model.py @@ -0,0 +1,213 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Optional, Tuple + +import torch + +import deepspeed.comm as dist + +from ...allocator import empty_from +from ...inference_utils import ActivationType, DtypeEnum +from .. import * +from ...modules.configs import * +from ...modules.interfaces import * +from ...ragged import RaggedBatchWrapper + +from .container import FalconNonTransformerContainer, FalconTransformerContainer + + +class FalconInferenceModel(DSTransformerModelBase): + """ + Inference model implementation for ragged batching for Llama-2 models. + """ + + _non_transformer: Optional[FalconNonTransformerContainer] + """ + Embed + unembed container. Specializing the type annotation. + """ + + _transformer: Optional[Iterable[FalconTransformerContainer]] + """ + Per-layer transformer container. Specializing the type annotation. + """ + """ + Properties inherited from `DSInferenceModelBase` + """ + + @property + def max_sequence_length(self) -> int: + return self._config.max_seq_length + + """ + Properties inherited from `DSTransformerModelBase` + """ + + @property + def num_layers(self) -> int: + return self._config.num_hidden_layers + + @property + def model_dim(self) -> int: + return self._config.hidden_size + + @property + def vocab_size(self) -> int: + return self._config.vocab_size + + @property + def head_size(self) -> int: + return self.model_dim // self.n_heads + + @property + def n_heads(self) -> int: + return self._config.num_attention_heads + + @property + def intermediate_dim(self) -> int: + return 4 * self._config.hidden_size + + @property + def n_heads_kv(self) -> int: + return self._config.num_kv_heads if (self._config.new_decoder_architecture + or not self._config.multi_query) else 1 + + @property + def activation_dtype(self) -> DtypeEnum: + if self._config.torch_dtype == torch.float16: + return DtypeEnum.fp16 + elif self._config.torch_dtype == torch.bfloat16: + return DtypeEnum.bf16 + else: + raise NotImplementedError("Only fp16 and bf16 are supported") + + @property + def mlp_activation_fn(self) -> ActivationType: + return ActivationType.GELU + + @property + def norm_type(self) -> NormTypeEnum: + return NormTypeEnum.LayerNorm + + @property + def positional_embedding_type(self) -> PositionalEmbeddingType: + return PositionalEmbeddingType.rotate_half + + @property + def positional_embedding_config(self) -> RotateHalfConfig: + """ + The positional embedding configuration for the model. + """ + return RotateHalfConfig() + + """ + Forward implementations + """ + + def _forward_embed(self, ragged_batch: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs the embedding lookup prior to running the transformer of the model. + + Arguments: + ragged_batch (RaggedBatchWrapper): The batch to embed. + + Returns: + torch.Tensor: The embedded batch. + """ + embed = self.embed(ragged_batch, self._non_transformer.word_emb) + + if embed.shape[-1] != self.model_dim: + raise ValueError(f"Embedding output shape {embed.shape} does not match model_dim {self.model_dim}") + + return embed + + def _forward_transformer_layer(self, layer_idx: int, residual: torch.Tensor, hidden_states: torch.Tensor, + ragged_batch_info: RaggedBatchWrapper) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Executes one (slightly offset) layer of the transformer. This implementation does a peak-ahead + optimization to fuse the layer norm of the next layer into the current layer. + + Arguments: + layer_idx (int): The index of the layer to execute. + residual (torch.Tensor): The residual tensor from the previous layer. + hidden_states (torch.Tensor): The hidden states from the previous layer. This is the + hidden states after pre normalization. + ragged_batch_info (RaggedBatchWrapper): The batch metadata. + """ + assert self.config.parallel_attn, "Only parallel attention implementation is supported" + + cur_params = self._transformer[layer_idx] + kv_cache = self.state_manager.get_cache(layer_idx) + + attn_ln_out = hidden_states + attn_hidden_state = self.qkv(attn_ln_out, cur_params.qkv_w, b=None) + attn_hidden_state = self.attn(attn_hidden_state, kv_cache, ragged_batch_info) + attention_output = self.attn_out(attn_hidden_state, cur_params.attn_out_w, b=None) + + if self.config.new_decoder_architecture: + residual, mlp_ln_out = self.norm(residual, + None, + gamma=cur_params.ln_mlp_gamma, + beta=cur_params.ln_mlp_beta) + else: + mlp_ln_out = hidden_states + + mlp_hidden_state = self.mlp_1(mlp_ln_out, cur_params.mlp_1_w, b=None) + mlp_output = self.mlp_2(mlp_hidden_state, cur_params.mlp_2_w, b=None) + + mlp_output.add_(attention_output) + + if self.tp_size > 1: + dist.all_reduce(mlp_output, group=self._base_mp_group) + + if layer_idx != self.num_layers - 1: + next_params = self._transformer[layer_idx + 1] + residual, mlp_output = self.norm(residual, + mlp_output, + next_params.ln_attn_gamma, + beta=next_params.ln_attn_beta) + else: + # On last layer, we just need to perform the residual add. Adding into the residual + # here is safe. + residual.add_(mlp_output) + + return residual, mlp_output + + def _forward_unembed(self, hidden_states: torch.Tensor, ragged_batch_info: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs unembedding of the hidden states to logits. This will only sample the final + token of each sequence. + """ + logits = self.unembed(hidden_states, + self._non_transformer.word_unembed, + ragged_batch_info, + gamma=self._non_transformer.final_norm_gamma, + beta=self._non_transformer.final_norm_beta) + + if self.tp_size > 1: + comm_buffer = empty_from(self._comm_logits, (self.tp_size, logits.shape[0], logits.shape[1])) + full_logits = empty_from(self._return_logits, (logits.shape[0], self.vocab_size)) + + dist.all_gather_into_tensor(comm_buffer, logits, group=self._base_mp_group) + + full_logits.copy_(comm_buffer.permute(1, 0, 2).reshape(logits.shape[0], self.vocab_size)) + + return full_logits + else: + return logits + + def forward(self, wrapped_batch: RaggedBatchWrapper) -> torch.Tensor: + residual = self._forward_embed(wrapped_batch) + + residual, hidden_states = self.norm(residual, + None, + gamma=self._transformer[0].ln_attn_gamma, + beta=self._transformer[0].ln_attn_beta) + + for layer_idx in range(self.num_layers): + residual, hidden_states = self._forward_transformer_layer(layer_idx, residual, hidden_states, + wrapped_batch) + + return self._forward_unembed(residual, wrapped_batch) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/policy.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/policy.py new file mode 100644 index 0000000000000000000000000000000000000000..c6612090a0df41228244c193b492493824a86394 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/falcon/policy.py @@ -0,0 +1,33 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any + +from ...config_v2 import RaggedInferenceEngineConfig +from ..inference_policy_base import ContainerMap, InferenceV2Policy +from .container import FalconNonTransformerContainer, FalconTransformerContainer +from .container import FalconNewArchTransformerContainer +from .model import FalconInferenceModel + + +class FalconPolicy(InferenceV2Policy): + + def instantiate_model(self, engine_config: RaggedInferenceEngineConfig, mp_group: Any) -> FalconInferenceModel: + return FalconInferenceModel(config=self._model_config, engine_config=engine_config, base_mp_group=mp_group) + + def build_container_map(self) -> ContainerMap: + map = ContainerMap() + + trans_container_cls = FalconNewArchTransformerContainer if self._model_config.new_decoder_architecture else FalconTransformerContainer + transformer_containers = [trans_container_cls(self.model) for _ in range(self.model.num_layers)] + + map.set_transformer_params(['transformer.h'], transformer_containers) + + map.set_non_transformer_params(FalconNonTransformerContainer(self.model)) + + map.set_unmapped_params( + [f'model.layers.{i}.self_attn.rotary_emb.inv_freq' for i in range(self.model.num_layers)]) + + return map diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/flat_model_helpers.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/flat_model_helpers.py new file mode 100644 index 0000000000000000000000000000000000000000..c5e02adaffc4d4f095c966ff31eb985bb27eca26 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/flat_model_helpers.py @@ -0,0 +1,282 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Dict, Iterable, Tuple, Optional +from os import path + +import torch + +from deepspeed.accelerator import get_accelerator +from deepspeed.ops.op_builder import RaggedUtilsBuilder +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from .layer_container_base import LayerContainer +from ..inference_parameter import InferenceParameter, STR_TO_DTYPE +from ..inference_utils import elem_size + + +def pad_to_aligned_offset(offset: int, alignment: int = 256) -> int: + """ + Pad the provided offset to a well-aligned value. + """ + return ((offset + alignment - 1) // alignment) * alignment + + +class TensorMetadata(DeepSpeedConfigModel): + """ + A class to represent a tensor specification. + """ + dtype: Optional[str] = None + shape: Optional[Tuple[int, ...]] = None + strides: Optional[Tuple[int, ...]] = None + offset: int + + +class ParameterMetadata(DeepSpeedConfigModel): + """ + A class to represent a parameter specification. + """ + core_param: Optional[TensorMetadata] = None + aux_params: Dict[str, TensorMetadata] = {} + + +class LayerMetadata(DeepSpeedConfigModel): + """ + A class to represent a layer specification. + """ + params: Dict[str, ParameterMetadata] = {} + + +class ModelMetadata(DeepSpeedConfigModel): + """ + A class to represent a model specification. + """ + policy: str = "" + layers: Dict[str, LayerMetadata] = {} + + +def make_param_filename(base: str, rank: int, n_ranks: int) -> str: + """ + Make a filename for a parameter file. + + Arguments: + rank: Rank of the file. + n_ranks: Total number of ranks. + + Returns: + str: Filename. + """ + return path.join(base, f"params_rank_{rank}_of_{n_ranks}.pt") + + +def make_metadata_filename(base: str, rank: int, n_ranks: int) -> str: + """ + Make a filename for a metadata file. + + Arguments: + rank: Rank of the file. + n_ranks: Total number of ranks. + + Returns: + str: Filename. + """ + return path.join(base, f"metadata_rank_{rank}_of_{n_ranks}.json") + + +def make_model_config_filename(base: str) -> str: + """ + Make a filename for a model config file. + + Arguments: + base: Base directory. + + Returns: + str: Filename. + """ + return path.join(base, "ds_model_config.json") + + +def flatten_inference_model( + transformer_containers: Iterable[LayerContainer], + non_transformer_container: LayerContainer, + policy_name: str, +) -> Tuple[torch.Tensor, ModelMetadata]: + """ + Flatten the underlying parameters into + + Arguments: + transformer_containers: Iterable of layer containers corresponding to the transformer + parameters. + non_transformer_container: Layer container corresponding to the non-transformer parameters. + policy_name: The name of the policy class (typically accessed with `type(policy).__name__`). + + Returns: + Iterable[Any]: Flattened list of parameters. + """ + alloc_fn = RaggedUtilsBuilder().load().allocate_view_on + + total_size = 0 + metadata = ModelMetadata(policy=policy_name) + + def process_layer(layer_container: LayerContainer, l_name: str, cur_offset: int) -> int: + """ + Iterate over the parameters of a single container and collect metadata for the final + flattened buffer. + + Arguments: + layer_container: The layer container to process. + l_name: The name of the layer container to key the metadata. + cur_offset: The current offset into the flattened buffer. + + Captured Variables: + metadata: The metadata object to populate. + + Returns: + int: The updated offset into the flattened buffer. + """ + try: + _ = layer_container.is_populated + except ValueError as e: + raise ValueError(f"Layer container {l_name} is not populated.") from e + + layer_metadata = LayerMetadata() + + for p_name in layer_container.annotation_attrs: + param = getattr(layer_container, p_name) + param_metadata = ParameterMetadata() + + if param is None: + param_metadata.core_param = TensorMetadata(offset=-1) + layer_metadata.params[p_name] = param_metadata + continue + + param_metadata.core_param = TensorMetadata(dtype=str(param.dtype), + shape=param.shape, + strides=param.stride(), + offset=cur_offset) + + cur_offset += pad_to_aligned_offset(elem_size(param.dtype) * param.numel()) + + for t_name, tensor in param.aux_attrs.items(): + param_metadata.aux_params[t_name] = TensorMetadata(dtype=str(tensor.dtype), + shape=tensor.shape, + strides=tensor.stride(), + offset=cur_offset) + + cur_offset += pad_to_aligned_offset(elem_size(tensor.dtype) * tensor.numel()) + + layer_metadata.params[p_name] = param_metadata + + metadata.layers[l_name] = layer_metadata + return cur_offset + + for i, layer in enumerate(transformer_containers): + l_name = f"transformer_layer_{i}" + total_size = process_layer(layer, l_name, total_size) + + l_name = "non_transformer" + total_size = process_layer(non_transformer_container, l_name, total_size) + + buffer = torch.empty(total_size, dtype=torch.uint8, device=get_accelerator().current_device()) + + def copy_layer(layer_container: LayerContainer, l_name: str) -> None: + """ + Local method for copying from the layer container to the flattened buffer. + + Arguments: + layer_container: The layer container to copy from. + l_name: The name of the layer container to key the metadata. + + Captured Variables: + buffer: The flattened buffer to copy into. + metadata: The metadata object to populate. + """ + l_metadata = metadata.layers[l_name] + for p_name in layer_container.annotation_attrs: + p_metadata = l_metadata.params[p_name] + param = getattr(layer_container, p_name) + + if param is None: + continue + + core_param = alloc_fn(param, buffer, p_metadata.core_param.offset) + core_param.copy_(param) + + aux_params = {} + + for t_name, tensor in param.aux_attrs.items(): + t_view = alloc_fn(tensor, buffer, p_metadata.aux_params[t_name].offset) + aux_params[t_name] = t_view + t_view.copy_(tensor) + + setattr(layer_container, p_name, InferenceParameter.initialize(core_param, **aux_params)) + + for i, layer in enumerate(transformer_containers): + l_name = f"transformer_layer_{i}" + copy_layer(layer, l_name) + + l_name = "non_transformer" + copy_layer(non_transformer_container, l_name) + + return buffer, metadata + + +def restore_inference_model(buffer: torch.Tensor, metadata: ModelMetadata, + transformer_containers: Iterable[LayerContainer], + non_transformer_container: LayerContainer) -> None: + """ + Restore the model from the buffer and metadata. + + Arguments: + buffer: Buffer containing the model parameters. + metadata: Metadata for the model. + transformer_containers: Iterable of transformer layer containers. + non_transformer_container: Non-transformer layer container. + """ + alloc_fn = RaggedUtilsBuilder().load().allocate_view_like + + def restore_layer(layer_container: LayerContainer, l_name: str) -> None: + """ + Local method for restoring a layer container from a flattened buffer. This + only constructs views for the parameters onto the buffer. No data movement + is performed. + + Arguments: + layer_container: The layer container to restore. + l_name: The name of the layer container to key the metadata. + + Captured Variables: + buffer: The flattened buffer to reconstruct views on top of. + metadata: The metadata object describing the each parameter in the model. + """ + l_metadata = metadata.layers[l_name] + + for p_name in layer_container.annotation_attrs: + p_metadata = l_metadata.params[p_name] + + if p_metadata.core_param.offset == -1: + layer_container.direct_injection(p_name, None) + continue + + dummy_tensor = torch.empty([], dtype=STR_TO_DTYPE[p_metadata.core_param.dtype]) + core_param = alloc_fn(p_metadata.core_param.shape, p_metadata.core_param.strides, dummy_tensor, buffer, + p_metadata.core_param.offset) + + aux_params = {} + + for t_name, t_metadata in p_metadata.aux_params.items(): + dummy_tensor = torch.empty([], dtype=STR_TO_DTYPE[t_metadata.dtype]) + t_view = alloc_fn(t_metadata.shape, t_metadata.strides, dummy_tensor, buffer, t_metadata.offset) + + aux_params[t_name] = t_view + + restored_param = InferenceParameter.initialize(core_param, **aux_params) + layer_container.direct_injection(p_name, restored_param) + + for i, layer in enumerate(transformer_containers): + l_name = f"transformer_layer_{i}" + restore_layer(layer, l_name) + + l_name = "non_transformer" + restore_layer(non_transformer_container, l_name) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/inference_model_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/inference_model_base.py new file mode 100644 index 0000000000000000000000000000000000000000..894a4137407e9b0c7c38f18d504cebf2977649bd --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/inference_model_base.py @@ -0,0 +1,272 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from abc import ABC, abstractmethod +from typing import Iterable, Optional, Tuple, Type + +import torch + +import deepspeed.comm as dist +from ..ragged import DSStateManager, RaggedBatchWrapper +from ..ragged.manager_configs import KVCacheConfig +from ..ragged import DSSequenceDescriptor +from ..model_implementations.layer_container_base import LayerContainer +from ..config_v2 import RaggedInferenceEngineConfig +from .flat_model_helpers import ModelMetadata + +try: + from functools import cached_property +except ImportError: + + def cached_property(func): + return property(func) + + +""" +This abstract class defines the interfaces that a model implementation should implement +in order to include anything that may be called by the engine. Most models should be able +to inherit from `DSInferenceTransformerModelBase` to reduce implementation work so it is recommended +to begin there. +""" +""" +Placeholder for typing the model config, which can vary based on model implementation/ +""" +DSModelImplementationConfig = Type['DSModelImplementationConfig'] +""" +Placeholder for typing the distributed comm object. + +TODO(cmikeh2): Replace when we have a more defined API for the inference communication system. +""" +MPType = Type["MPType"] + + +class DSInferenceModelBase(torch.nn.Module, ABC): + """ + Implementation of a model for inference composable with ragged batching. + """ + + _config: DSModelImplementationConfig + """ + Model-specific configuration. No abstraction surrounds this yet. + """ + + _engine_config: RaggedInferenceEngineConfig + """ + Engine configuration. + """ + + _base_mp_group: MPType + """ + Base communication group for Tensor-parallel inference. + """ + + _non_transformer: Optional[LayerContainer] + """ + Abstract container for storing both embedding (pre-transformer) and unembedding (post-transformer) + parameters. This attribute should be None at model instantiation until the Policy sets + the model parameters. These parameters are grouped together since many model implementations + will tie the embedding and unembedding parameters together. + """ + + _transformer: Optional[Iterable[LayerContainer]] + """ + List of abstract containers (1 per layer) for storing transformer (transformer) + parameters. This attribute should be None at model instantiation until the Policy + sets the model parameters. + """ + + state_manager: Optional[DSStateManager] + """ + Since the state manager is lazy initialized, by the engine, it is not guaranteed to be present + until full initialization. + """ + + def __init__(self, config: DSModelImplementationConfig, engine_config: RaggedInferenceEngineConfig, + base_mp_group: MPType) -> None: + """ + Minimal initialization of the model. + + Arguments: + config (DSModelImplementationConfig): Model-specific configuration. No assumptions + should be made about this config that are not closely tied to the specific + model implementation. + engine_config (RaggedInferenceEngineConfig): Engine configuration. + base_mp_group (MPType): Base communication group for Tensor-parallel inference. + """ + super().__init__() + self._config = config + self._engine_config = engine_config + self._base_mp_group = base_mp_group + + # Set to None until the Policy sets the model parameters + self._non_transformer = None + self._transformer = None + self._flattened_param_buffer = None + self._flattened_param_metadata = None + + @property + def config(self) -> DSModelImplementationConfig: + """ + The model config. + """ + return self._config + + def set_parameters(self, transformer: Iterable[LayerContainer], non_transformer: LayerContainer, + flattened_param_buffer: torch.Tensor, flattened_param_metadata: ModelMetadata): + """ + Set the model parameters for the embedding, transformer, and unembedding containers. + """ + self._transformer = transformer + self._non_transformer = non_transformer + self._flattened_param_buffer = flattened_param_buffer + self._flattened_param_metadata = flattened_param_metadata + + def set_state_manager(self, state_manager: DSStateManager): + """ + Sets the state manager attribute. This is called by the inference engine after + the model is fully initialized. + """ + self.state_manager = state_manager + + @cached_property + def tp_rank(self) -> int: + """ + The rank of the current process. + + # TODO(cmikeh2): Kind of a hack right now, but this is too verbose to use at + the frequency we need. + """ + return dist.get_rank(group=self._base_mp_group) + + @cached_property + def tp_size(self) -> int: + """ + The total number of processes. + + # TODO(cmikeh2): Kind of a hack right now, but this is too verbose to use at + the frequency we need. + """ + return dist.get_world_size(group=self._base_mp_group) + + @property + def model_config(self): + """ + The model config. + """ + return self._config + + @property + def engine_config(self): + """ + The engine config. + """ + return self._engine_config + + @property + def flattened_params(self) -> Optional[torch.Tensor]: + """ + The flattened parameter buffer. + """ + return self._flattened_param_buffer + + @property + def flattened_param_metadata(self) -> Optional[ModelMetadata]: + """ + The flattened parameter metadata. + """ + return self._flattened_param_metadata + + @abstractmethod + def get_kv_requirements(self, sequence: DSSequenceDescriptor, max_new_tokens: int, + max_new_blocks: Tuple[int, ...]) -> Tuple[int, torch.Tensor]: + """ + Given a sequence and the number of new tokens in the sequence, determine the + number of new KV blocks needed to support the sequence. This method is + used to help the engine provide schedulability APIs and can be used as a helper + for ``maybe_allocate_kv``. + + Args: + sequence (DSSequenceDescriptor): The sequence for which to allocate KV-storage. + max_new_tokens (int): Maximum number of tokens to hypothetically schedule. + max_new_blocks (int): Maximum number of blocks to hypothetically allocate. + + Returns: + Tuple[int, torch.Tensor]: The tuple of number of tokens scheduled and number + of blocks allocated (per KV cache). In general, only one of these numbers will + match the corresponding input argument, but this is not guaranteed. + """ + raise NotImplementedError() + + @abstractmethod + def get_remaining_block_capacity(self, sequence: DSSequenceDescriptor) -> int: + raise NotImplementedError() + + @abstractmethod + def maybe_allocate_kv(self, sequence: DSSequenceDescriptor, n_new_tokens: int) -> None: + """ + Given a sequence and the number of new tokens in the sequence, determine + whether or not additional KV-storage is needed and allocate it if so. + + Args: + sequence (DSSequenceDescriptor): The sequence for which to allocate KV-storage. + n_new_tokens (int): The number of new tokens in the sequence. + """ + raise NotImplementedError() + + @abstractmethod + def kv_cache_config(self) -> Tuple[KVCacheConfig, ...]: + """ + Return the KV-cache configuration for this model. This should be a tuple of one or more + KVCacheConfig objects (one for each distinct cache group). + """ + raise NotImplementedError() + + @property + @abstractmethod + def max_sequence_length(self) -> int: + """ + The maximum sequence length supported by the model. + """ + ... + + def maybe_free_kv(self, sequence: DSSequenceDescriptor) -> None: + """ + After completing a forward pass, determine whether or not the there are any KV blocks + that maybe freed since they are no longer in use. + + Consider the following example: + + We have a block size of 4 and a local window size of 8. At the beginning of the forward + pass there 10 tokens had been seen and the new forward has a size of 4. This would lend + itself to the following cache structure prior to the forward: + [[0, 1, 2*, 3*] [4*, 5*, 6*, 7*] [8*, 9*, x, x] [x x x x]] + Where x's denote empty cache locations and * denote values that are needed for attention + of the next open slot. After the forward, the cache would look like the following: + [[0, 1, 2, 3] [4, 5, 6*, 7*] [8*, 9*, 10*, 11*] [12* 13* x x]] + In this case, the first block is no longer needed since it is not needed for any future + local attention windows. This function would be responsible for freeing that block. + + Default behavior assumes no local patterns that require freeing and in general should + be sufficient. + """ + pass + + @abstractmethod + def prepare_batch(self, wrapped_batch: RaggedBatchWrapper) -> None: + """ + This will be called before each forward with the intent of building forward-specific metadata + about a batch. The intent here is to build data structures like attention atoms without necessarily + needing to implement graphable kernels to do so. + + Abstract so as to force model implementations to opt out of doing anything here explicitly. + """ + raise NotImplementedError() + + def forward(wrapped_batch: RaggedBatchWrapper) -> torch.Tensor: + """ + Complete a forward pass of the model. This interface should be graphable, so it + should not rely on the ability to use python control flow. + """ + raise NotImplementedError() diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/inference_policy_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/inference_policy_base.py new file mode 100644 index 0000000000000000000000000000000000000000..2f4266a8cb880ccf7ce97a82c708962d3b75b059 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/inference_policy_base.py @@ -0,0 +1,220 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import json +from abc import ABC, ABCMeta, abstractmethod +from typing import Any, Iterable, List, Optional, Union + +import torch + +from ..config_v2 import RaggedInferenceEngineConfig +from ..checkpoint import CheckpointEngineBase +from ..logging import inference_logger +from .layer_container_base import LayerContainer +from .inference_model_base import DSInferenceModelBase +from .flat_model_helpers import ( + flatten_inference_model, + make_param_filename, + make_metadata_filename, + ModelMetadata, + restore_inference_model, +) + +POLICIES = {} + + +class ContainerMap: + + def __init__(self) -> None: + self._prefix_map = {} + self._transformer_params = None + self._non_transformer_params = None + + @property + def transformer_params(self) -> Iterable[LayerContainer]: + return self._transformer_params + + @property + def non_transformer_params(self) -> LayerContainer: + return self._non_transformer_params + + def set_transformer_params(self, prefixes: Union[str, Iterable[str]], containers: List[LayerContainer]) -> None: + if not isinstance(containers, list): + raise ValueError( + f"The transformer containers should be a list, of one container per layer, but got {type(containers)} instead." + ) + + self._transformer_prefixes = prefixes if isinstance(prefixes, list) else [prefixes] + self._transformer_params = containers + + def set_non_transformer_params(self, container: LayerContainer) -> None: + self._non_transformer_params = container + + def set_unmapped_params(self, prefixes: Union[str, Iterable[str]]) -> None: + self._unmapped_prefixes = prefixes + + def map_param(self, name, parameter) -> None: + for unmapped_prefix in self._unmapped_prefixes: + if name.startswith(unmapped_prefix): + inference_logger().debug(f"Ignoring: {name} for {unmapped_prefix}") + return + + for transformer_prefix in self._transformer_prefixes: + if name.startswith(transformer_prefix): + popped_name = name[len(transformer_prefix) + 1:] + layer_idx = popped_name.split(".")[0] + assert layer_idx.isdigit( + ), f"expected name to start w. list index but got {layer_idx} instead, name={name}" + layer_idx = int(layer_idx) + inference_logger().debug( + f"Setting: {'.'.join(popped_name.split('.')[1:])} for layer-idx={layer_idx} to {parameter.shape}") + self._transformer_params[layer_idx].set_dependency(".".join(popped_name.split(".")[1:]), parameter) + return + + try: + inference_logger().debug(f"Setting: {name} to {parameter.shape}") + self._non_transformer_params.set_dependency(name, parameter) + except ValueError: + # Catch the ValueError here from the non_transformer_params because we are knowingly + # calling it with something that may not match. This should allow us to raise a slightly more + # informative error message. + raise ValueError(f"Cannot find container for {name}, please double check the Containers/ContainerMap") + + def validate(self) -> None: + if not self._non_transformer_params.is_initialized: + raise RuntimeError("Non-transformer parameters not fully initialized after checkpoint load.") + + for layer_idx, container in enumerate(self._transformer_params): + if not container.is_initialized: + raise RuntimeError( + f"Transformer container at index {layer_idx} not fully initialized after checkpoint load.") + + +class PolicyMeta(ABCMeta): + + def __new__(cls, name, bases, dct): + new_obj = super().__new__(cls, name, bases, dct) + if name != "InferenceV2Policy": + POLICIES[name] = new_obj + return new_obj + + +class InferenceV2Policy(ABC, metaclass=PolicyMeta): + """ + The InferenceV2Policy is the base class for all inference policies. An inference policy + is responsible for instantiating the inference model and mapping the parameters from the + checkpoint engine to the model itself. + """ + + def __init__( + self, + model_config: Any, + checkpoint_engine: Optional[CheckpointEngineBase] = None, + inf_checkpoint_path: Optional[str] = None, + ) -> None: + """ + Create the Policy with sufficient context to build the model. There are two supported + model creation mechanisms. + + The first is the generalized ``checkpoint_engine`` which + will iterate over the parameters of the model and provide them to the policy. These in + turn will be sharded/transformed by the model implementation. + + The second is used to re-create a previously serialized DeepSpeed inference model. These + checkpoints should not be used across different model backend configurations. + + TODO(cmikeh2): Enforce this in code + """ + if checkpoint_engine is None and inf_checkpoint_path is None: + raise ValueError("Either checkpoint_engine or ds_checkpoint_path must be provided.") + + if checkpoint_engine is not None and inf_checkpoint_path is not None: + raise ValueError("Only one of checkpoint_engine or ds_checkpoint_path can be provided.") + + self._checkpoint_engine = checkpoint_engine + self._inf_checkpoint_path = inf_checkpoint_path + self._model_config = model_config + + def build_model(self, engine_config: RaggedInferenceEngineConfig, mp_group: Any) -> DSInferenceModelBase: + """ + Completely instantiate the inference model. This will both create the ops needed to run the + model, as well as load the model parameters via the checkpoint engine. For more context + on each of these components please see ``instantiate_model`` and ``populate_model_parameters``. + + Arguments: + engine_config: The config that has been used to instantiate the engine. This is used + to communicate to the model implementation the limits on batches (sequences/tokens) + and bound the size of intermediate buffers. + mp_group: Object to enable communication between tensor parallel ranks. + + Returns: + DSInferenceModelBase: An implementation of the inference model abstraction that will be + run by the engine. + """ + self.model = self.instantiate_model(engine_config, mp_group) + self.populate_model_parameters() + return self.model + + @abstractmethod + def instantiate_model(self, engine_config: RaggedInferenceEngineConfig) -> DSInferenceModelBase: + """ + Instantiate the inference model. Depending on the engine/model config, this could be where + different model implementations could be selected. + + Arguments: + engine_config: The config that has been used to instantiate the engine. This is used + to communicate to the model implementation the limits on batches (sequences/tokens) + and bound the size of intermediate buffers. + + Returns: + DSInferenceModelBase: An implementation of the inference model abstraction that will be + run by the engine. + """ + ... + + @abstractmethod + def build_container_map(self) -> ContainerMap: + """ + Build a dictionary representing the structure of the string prefixes leading + to the parameters to be mapped to the container. + + Returns: + ContainerMap: An instantiated mapping describing how checkpoint prefixes map + to ``LayerContainer`` instances. + """ + raise NotImplementedError() + + def populate_model_parameters(self) -> None: + """ + This model will iterate over the parameters (as provided by the checkpoint engine) and + use the container map built by ``build_container_map`` to populate the model + """ + + container_map = self.build_container_map() + + if self._checkpoint_engine is not None: + for name, parameter in self._checkpoint_engine.parameters(): + container_map.map_param(name, parameter) + + buffer, metadata = flatten_inference_model(container_map.transformer_params, + container_map.non_transformer_params, self.__class__.__name__) + else: + + buffer_path = make_param_filename(self._inf_checkpoint_path, self.model.tp_rank, self.model.tp_size) + metadata_path = make_metadata_filename(self._inf_checkpoint_path, self.model.tp_rank, self.model.tp_size) + + buffer = torch.load(buffer_path, weights_only=False) + metadata = json.load(open(metadata_path, "r")) + metadata = ModelMetadata.parse_raw(metadata) + + restore_inference_model(buffer, metadata, container_map.transformer_params, + container_map.non_transformer_params) + + container_map.validate() + + self.model.set_parameters(transformer=container_map.transformer_params, + non_transformer=container_map.non_transformer_params, + flattened_param_buffer=buffer, + flattened_param_metadata=metadata) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/inference_transformer_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/inference_transformer_base.py new file mode 100644 index 0000000000000000000000000000000000000000..fae67dc8fc2ad807df5ed4b337bd709cab20c9b5 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/inference_transformer_base.py @@ -0,0 +1,617 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from abc import abstractmethod +from typing import Optional + +import torch + +from deepspeed.accelerator import get_accelerator +from ..config_v2 import RaggedInferenceEngineConfig +from ..inference_utils import ActivationType, ceil_div, is_gated +from ..model_implementations import * +from ..model_implementations.sharding import * +from ..modules.configs import ( + DSEmbeddingsConfig, + DSLinearConfig, + DSMoEConfig, + DSNormConfig, + DSSelfAttentionConfig, + DSUnembedConfig, + NormTypeEnum, + PositionalEmbeddingType, + RotateHalfConfig, +) +from ..modules import heuristics +from ..ragged import ( + DSSequenceDescriptor, + KVCacheConfig, + RaggedBatchWrapper, +) +from .inference_model_base import ( + DSInferenceModelBase, + DSModelImplementationConfig, + MPType, +) +from ..inference_parameter import InferenceParameter + +try: + from functools import cached_property +except ImportError: + + def cached_property(func): + return property(func) + + +class DSTransformerModelBase(DSInferenceModelBase): + """ + Dimensioning properties + """ + + @property + @abstractmethod + def num_layers(self) -> int: + """ + Number of the layers in the model + """ + ... + + @property + @abstractmethod + def model_dim(self) -> int: + """ + Size of embedding projection and residuals. + """ + ... + + @property + @abstractmethod + def vocab_size(self) -> int: + """ + Size of the vocabulary (including padding). + """ + ... + + @property + @abstractmethod + def head_size(self) -> int: + """ + Size of each attention head. + """ + ... + + @property + @abstractmethod + def n_heads(self) -> int: + """ + The number of query heads on the model. This should not take into account + any dimension reductions from model sharding. + """ + ... + + @property + def n_heads_q(self) -> int: + """ + Alias to n_heads. + """ + return self.n_heads + + @property + def n_heads_kv(self) -> int: + """ + The number of key and value heads on the model. For GQA or MQA, overload this attribute. + Otherwise it adopts MHA formulations and uses n_heads. This should not take into account + any dimension reductions from model sharding. + """ + return self.n_heads + + @property + @abstractmethod + def intermediate_dim(self) -> int: + """ + The size of the (unsharded) intermediate projection dim. For a gated activation function + this is the size of the input to the second MLP layer. This should not take into account + any dimension reductions from model sharding. + """ + ... + + @property + @abstractmethod + def positional_embedding_type(self) -> PositionalEmbeddingType: + """ + The type of positional embedding used by the model. + """ + ... + + """ + Architectural properties + """ + + @property + @abstractmethod + def activation_dtype(self) -> torch.dtype: + """ + The activation dtype of the model. + """ + ... + + @property + @abstractmethod + def mlp_activation_fn(self) -> ActivationType: + """ + The activation function used in the MLP. + """ + ... + + @property + @abstractmethod + def norm_type(self) -> NormTypeEnum: + """ + The type of normalization used in the model. + """ + ... + + @property + @abstractmethod + def positional_embedding_config(self) -> Optional[RotateHalfConfig]: + """ + The positional embedding configuration for the model. + """ + ... + + """ + Derived helpers + """ + + @cached_property + def n_heads_q_local(self) -> int: + """ + Number of local heads post sharding. + """ + return get_local_heads(self.tp_rank, self.tp_size, self.n_heads_q, self.n_heads_kv)[0] + + @cached_property + def n_heads_kv_local(self) -> int: + """ + Number of local heads post sharding. + """ + return get_local_heads(self.tp_rank, self.tp_size, self.n_heads_q, self.n_heads_kv)[1] + + @property + def gated_mlp(self) -> bool: + """ + Return a boolean to determine whether the model uses a gated activation function. + """ + return is_gated(self.mlp_activation_fn) + + """ + Method implementations + """ + + def __init__(self, config: DSModelImplementationConfig, engine_config: RaggedInferenceEngineConfig, + base_mp_group: MPType) -> None: + """ + Base implementation for initialization. By default, this will initialize + the traditional components of a transformer model: + - Embedding + - QKV projection + - Self attention + - Attention output projection + - Feed forward network + - Normalization + - Unembedding + + Arguments: + config (DSModelImplementationConfig): Model-specific configuration. No assumptions + should be made about this config that are not closely tied to the specific + model implementation. + engine_config (RaggedInferenceEngineConfig): Engine configuration. + base_mp_group (MPType): Base communication group for Tensor-parallel inference. + """ + super().__init__(config, engine_config, base_mp_group) + + self.make_norm_layer() + self.make_qkv_layer() + self.make_attn_layer() + self.make_attn_out_layer() + self.make_mlp_1_layer() + self.make_mlp_2_layer() + self.make_embedding_layer() + self.make_unembedding_layer() + self._kv_cache_config = None + + ######### Embedding ######### + def make_embedding_layer(self) -> None: + """ + Performs setup and creates embedding DSModule. This will set the `self.embed` attribute. + """ + + embed_config = DSEmbeddingsConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + residual_dtype=self.activation_dtype, + embedding_dim=self.model_dim, + ) + + self.embed = heuristics.instantiate_embed(embed_config, self._engine_config) + + def transform_embedding_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Performs embedding sharding along the channels dimension. + """ + # Until we can do non-contiguous all-gather, we won't shard the embedding parameters. + param = param.to(self.activation_dtype.value) + return InferenceParameter.initialize(param) + + ######### Unembedding ######### + def make_unembedding_layer(self) -> None: + """ + Performs setup and creates an unembedding layer. This implementation assumes + normalization prior to the LM head projection. If this does not match the model's + implementation, override this method. This will set the ``self.unembed`` attribute. + """ + unembed_dim = sharded_unembed_dim(self.vocab_size, self.tp_rank, self.tp_size) + + unembed_config = DSUnembedConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + max_sequences=self._engine_config.state_manager.max_ragged_sequence_count, + dtype=self.activation_dtype, + model_dim=self.model_dim, + vocab_size=unembed_dim, + norm_type=self.norm_type, + ) + + self.unembed = heuristics.instantiate_unembed(unembed_config, self._engine_config) + + if self.tp_size > 1: + self._comm_logits = torch.empty(self.tp_size, + self._engine_config.state_manager.max_ragged_sequence_count, + unembed_dim, + device=get_accelerator().current_device(), + dtype=self.activation_dtype.value) + self._return_logits = torch.empty(self._engine_config.state_manager.max_ragged_sequence_count, + self.vocab_size, + device=get_accelerator().current_device(), + dtype=self.activation_dtype.value) + + def transform_unembed_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Performs sharding along the vocab dimension. + """ + param = shard_unembed_param(param, self.tp_rank, self.tp_size).to(self.activation_dtype.value) + return InferenceParameter.initialize(param) + + ######### QKV ######### + def make_qkv_layer(self) -> None: + """ + Instantiates the linear projection layer for the QKV linear layer. This sets the + `self.qkv` attribute. + """ + out_features = qkv_out_features(self.model_dim, self.tp_rank, self.tp_size, self.head_size, self.n_heads_q, + self.n_heads_kv) + + linear_config = DSLinearConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + in_channels=self.model_dim, + out_channels=out_features, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + ) + + self.qkv = heuristics.instantiate_linear(linear_config, self._engine_config) + + def transform_qkv_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Passes a QKV parameter to the underlying implementation for any necessary + transformations. + + Args: + param (torch.Tensor): The parameter to transform. This may be either a bias or weight and should have + the shape (out_neurons, in_neurons) + """ + param = shard_qkv_param(param, self.tp_rank, self.tp_size, self.head_size, self.n_heads_q, self.n_heads_kv) + return self.qkv.transform_param(param) + + ######### Attention ######### + def make_attn_layer(self) -> None: + """ + Builds the attention layer for the model. This sets the `self.attn` attribute. + """ + softmax_scale = 1.0 / (self.head_size**0.5) + + attn_config = DSSelfAttentionConfig(max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + n_heads_q=self.n_heads_q_local, + n_heads_kv=self.n_heads_kv_local, + head_size=self.head_size, + max_sequences=self._engine_config.state_manager.max_ragged_sequence_count, + scale_factor=softmax_scale, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + positional_embedding_type=self.positional_embedding_type, + positional_embedding_config=self.positional_embedding_config) + + self.attn = heuristics.instantiate_attention(attn_config, self._engine_config) + + def get_kv_requirements(self, sequence: DSSequenceDescriptor, max_new_tokens: int, + max_new_blocks: int) -> Tuple[int, int]: + """ + See ``DSInferenceModelBase.get_kv_requirements`` for documentation. + + This method assumes an autoregressive dense attention pattern. Override this method + if this does not match the model's attention pattern. + """ + total_tokens = sequence.seen_tokens + max_new_tokens + req_blocks = ceil_div(total_tokens, self.attn.kv_block_size) + block_lim = req_blocks - sequence.cur_allocated_blocks + + if block_lim <= max_new_blocks: + return max_new_tokens, block_lim + + token_capacity = (max_new_blocks + + sequence.cur_allocated_blocks) * self.attn.kv_block_size - sequence.seen_tokens + + return token_capacity, max_new_blocks + + def get_remaining_block_capacity(self, sequence: DSSequenceDescriptor) -> int: + return sequence.seen_tokens % self.attn.kv_block_size + + def maybe_allocate_kv(self, sequence: DSSequenceDescriptor, n_new_tokens: int) -> None: + """ + See ``DSInferenceModelBase.maybe_allocate_kv`` for documentation. + + This method assumes an autoregressive dense attention pattern. Override this method + if this does not match the model's attention pattern. + """ + free_block = self.state_manager.free_blocks[0] + _, n_needed_blocks = self.get_kv_requirements(sequence, n_new_tokens, free_block) + + if n_needed_blocks > 0: + new_blocks = self.state_manager.allocate_blocks(n_needed_blocks) + sequence.extend_kv_cache(new_blocks) + + def kv_cache_config(self) -> Tuple[KVCacheConfig, ...]: + """ + See ``DSInferenceModelBase.kv_cache_config`` for documentation. + + This method assumes an autoregressive dense attention pattern. Override this method + if this does not match the model's attention pattern. + """ + if self._kv_cache_config is None: + cache_shape = (self.num_layers, self.n_heads_kv_local, self.head_size) + max_blocks = ceil_div(self.max_sequence_length, self.attn.kv_block_size) + self._kv_cache_config = KVCacheConfig(block_size=self.attn.kv_block_size, + cache_shape=cache_shape, + cache_dtype=self.activation_dtype, + max_blocks_per_allocation_group=max_blocks) + return (self._kv_cache_config, ) + + def prepare_batch(self, wrapped_batch: RaggedBatchWrapper) -> None: + """ + See ``DSInferenceModelBase.prepare_batch`` for documentation. + + This method assumes an autoregressive dense attention pattern. Override this method + if this does not match the model's attention pattern. + """ + self.attn.build_atoms(wrapped_batch) + + ######### Attention output ######### + def make_attn_out_layer(self) -> None: + """ + Instantiates the linear projection layer for the attention output linear layer. This sets the + `self.attn_out` attribute. + """ + in_features = attn_out_in_features(self.model_dim, self.tp_rank, self.tp_size, self.head_size, self.n_heads_q, + self.n_heads_kv) + + linear_config = DSLinearConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + in_channels=in_features, + out_channels=self.model_dim, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + ) + + self.attn_out = heuristics.instantiate_linear(linear_config, self._engine_config) + + def transform_attn_out_param(self, param: torch.Tensor) -> Optional[InferenceParameter]: + """ + Shards an attention output projection parameter and passes it to the underlying + implementation for any necessary transformations. This will return `None` for bias parameters + if they are not on TP rank 0. + + Args: + param (torch.Tensor): The parameter to transform. This may be either a bias or weight and should have + the shape (out_neurons, in_neurons). + """ + param = shard_attn_out_param(param, self.tp_rank, self.tp_size, self.head_size, self.n_heads_q, + self.n_heads_kv) + + if param is not None: + param = self.attn_out.transform_param(param) + + return param + + ######### MLP ######### + def make_mlp_1_layer(self) -> None: + """ + Instantiates the linear projection layer for the first MLP in the feedforward network. + This sets the `self.mlp_1` attribute. + """ + shard_size = sharded_intermediate_dim(self.intermediate_dim, self.tp_size, self.tp_rank) + + linear_config = DSLinearConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + in_channels=self.model_dim, + out_channels=shard_size, + activation=self.mlp_activation_fn, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + ) + + self.mlp_1 = heuristics.instantiate_linear(linear_config, self._engine_config) + + def transform_mlp_1_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Shards the first MLP parameter and passes it to the underlying implementation + for any necessary transformations. + + Args: + param (torch.Tensor): The parameter to transform. This may be either a bias or weight and should have + the shape (out_neurons, in_neurons). + """ + param = shard_mlp_1_param(param, self.tp_rank, self.tp_size, gated=self.gated_mlp) + + return self.mlp_1.transform_param(param) + + def make_mlp_2_layer(self) -> None: + """ + Instantiates the linear projection layer for the second MLP in the feedforward network. + This sets the `self.mlp_2` attribute. + """ + shard_size = sharded_intermediate_dim(self.intermediate_dim, self.tp_size, self.tp_rank) + + linear_config = DSLinearConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + in_channels=shard_size, + out_channels=self.model_dim, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + ) + + self.mlp_2 = heuristics.instantiate_linear(linear_config, self._engine_config) + + def transform_mlp_2_param(self, param: torch.Tensor) -> Optional[InferenceParameter]: + """ + Shards the second MLP parameter and passes it to the underlying implementation + for any necessary transformations. This will return `None` for bias parameters + if they are not on TP rank 0. + + Args: + param (torch.Tensor): The parameter to transform. This may be either a bias or weight and should have + the shape (out_neurons, in_neurons). + """ + param = shard_mlp_2_param(param, self.tp_rank, self.tp_size) + + if param is not None: + param = self.mlp_2.transform_param(param) + + return param + + ######### Norm ######### + def make_norm_layer(self) -> None: + """ + Instantiates the normalization layer for the model. This sets the `self.norm` attribute. + + TODO(cmikeh2): In the future we'll distinguish between the different norm objects, + but for now we'll just use the same one for all of them. + """ + norm_config = DSNormConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + type=self.norm_type, + channels=self.model_dim, + residual_dtype=self.activation_dtype, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + ) + + self.norm = heuristics.instantiate_pre_norm(norm_config, self._engine_config) + + def transform_norm_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Passes a normalization parameter to the underlying implementation for any + necessary transformations. + + TODO(cmikeh2): In the future we'll distinguish between the different norm objects, + but for now we'll just use the same one for all of them. + + Args: + param (torch.Tensor): The parameter to transform. This may be either a bias or weight and should have + shape (model_dim,) + """ + return self.norm.transform_param(param) + + +class DSMoETransformerModelBase(DSTransformerModelBase): + + @property + def n_experts(self) -> int: + """ + Return the number of experts in the model. + """ + raise NotImplementedError("Attempted to access an unimplemented number of experts") + + @property + def n_top_k(self) -> int: + """ + Number of experts per token. + """ + raise NotImplementedError("Attempted to access an unimplemented number of experts per token") + + @property + def normalize_expert_scores(self) -> bool: + """ + Whether to normalize expert scores. If true, sum(expert_scores) = 1. + """ + raise NotImplementedError("Attempted to access an unimplemented normalization flag") + + def make_moe_layer(self) -> None: + """ + Instantiates the MoE layer for the model. This sets the `self.moe` attribute. + """ + sharded_dim = sharded_intermediate_dim(self.intermediate_dim, self.tp_size, self.tp_rank) + + moe_config = DSMoEConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + model_dim=self.model_dim, + intermediate_features=sharded_dim, + activation=self.mlp_activation_fn, + n_experts=self.n_experts, + top_k=self.n_top_k, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + normalize_scores=self.normalize_expert_scores, + ) + + self.moe = heuristics.instantiate_moe(moe_config, self._engine_config) + + def transform_moe_gate_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Passes a MoE gate parameter to the underlying implementation for any necessary transformations. + + TODO(cmikeh2): This will need to be updated/overridden for expert parallelism. + """ + return self.moe.transform_gate_param(param) + + def transform_moe_mlp_1_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Shards the first MoE param and passes it to the underlying implementation. Since it's possible for an architecture + to have both MoE and non-MoE layers, this can't be overloaded on the MLP1 transform. Furthermore, since both + the MoE DSModule owns both MLP1 and MLP2, under certain sharding conditions it's not possible for the model implementation + to infer from the shape whether to perform a different transformation based on MLP1 or MLP2. This (and the below) + separations are intended to solve both these issues. + + Args: + param (torch.Tensor): The parameter to transform. This should have shape (n_experts, out_neurons, in_neurons). + """ + param = shard_mlp_1_param(param, self.tp_rank, self.tp_size, gated=self.gated_mlp, is_moe=True) + + return self.moe.transform_moe_mlp_1_param(param) + + def transform_moe_mlp_2_param(self, param: torch.Tensor) -> Optional[torch.Tensor]: + """ + Shards the second MoE param and passes it to the underlying implementation. See the above for context on why this API + exists. + + This will return `None` for expert bias params not on TP rank 0. NOTE(cmikeh2): Does it make sense to round-robin assign? + My intuition is that this will make debugging much more difficult for minimal memory reduction. + + Args: + param (torch.Tensor): The parameter to transform. This should have shape (n_experts, out_neurons, in_neurons). + """ + param = shard_mlp_2_param(param, self.tp_rank, self.tp_size, is_moe=True) + + if param is not None: + param = self.moe.transform_moe_mlp_2_param(param) + + return param diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/layer_container_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/layer_container_base.py new file mode 100644 index 0000000000000000000000000000000000000000..feb65b4a5f5d143807bddb44ee09844aeeb00141 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/layer_container_base.py @@ -0,0 +1,355 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import re +from typing import Type + +import torch + +from deepspeed.accelerator import get_accelerator +from .parameter_base import ParameterBase, ParametrizedList +from ..inference_parameter import InferenceParameter + +# Currently have dependency loops for the type hints. +InferenceModel = Type["InferenceModel"] +LayerContainer = Type["LayerContainer"] # noqa: F811 + +MAPPING_KEY = "PARAM_MAPPING" +PLIST_HELPERS = "_ds_plist_strip_vals" + + +def make_finalization_callback(all_names: str): + """ + Helper method for building the finalization callback for a LayerContainer. This + is not client code and should not be used or called directly. + """ + + def finalization_callback(self, param: ParameterBase, finalized_param: torch.Tensor) -> None: + """ + Callback for when a parameter is finalized. + """ + self._finalized_params += 1 + + for name in all_names: + if getattr(self, name) is param: + setattr(self, name, finalized_param) + + return finalization_callback + + +class LayerMetaclass(type): + """ + MetaClass for the LayerContainer base class. This class will parse the annotations + of the class that correspond to `ParameterBase` and create None initializers for each + as well as a finalization callback that for when each `ParameterBase` is finalized + and should be replaced with a Tensor. + """ + + def __new__(cls, clsname, bases, attrs): + + annotations = attrs.get("__annotations__", {}) + + for base in bases: + # We'll pick up all annotations on any base classes. This will allow us to + # to use inheritance to share common parameter groups in base classes. + if hasattr(base, "__annotations__"): + annotations.update(base.__annotations__) + + if hasattr(base, MAPPING_KEY): + if MAPPING_KEY not in attrs: + # This is likely a fail state. If a parent has MAPPING KEY but the child does + # not, then we're guaranteed only a subset of the parameters will be mapped. + attrs[MAPPING_KEY] = {} + attrs[MAPPING_KEY].update(getattr(base, MAPPING_KEY)) + + all_names = [name for name, annotation in annotations.items() if issubclass(annotation, ParameterBase)] + + if MAPPING_KEY in attrs: + # If we have a mapping key at all, then we will enter the validation mode for building + # helpers for mapping and ensuring we have complete mapping. + + # First we'll build a flat list of every dependency for this layer. + all_deps = set() + for name in all_names: + parameter_deps = [ + name for name, annotation in annotations[name].__annotations__.items() + if issubclass(annotation, (torch.Tensor, ParametrizedList)) + ] + + all_deps.update([f"{name}.{dep}" for dep in parameter_deps]) + + # Create static helper for doing the string processing only once. + attrs[PLIST_HELPERS] = [] + + # Iterate over all the mappings + for src_name, target_or_targets in attrs[MAPPING_KEY].items(): + if isinstance(target_or_targets, str): + target_or_targets = [target_or_targets] + + actual_targets = [] + for target_name in target_or_targets: + base_dependency, dependency_attr = target_name.split(".") + + # Check for invalid mappings + if base_dependency not in all_names: + raise ValueError( + "Target parameter \"{}\" not found in this layer. Valid targets are {}".format( + base_dependency, all_names)) + if dependency_attr not in annotations[base_dependency].__annotations__: + # This check is not universal (see below) if a single dependency is being + # mapped to by a single row. + raise ValueError( + "Target dependency \"{}\" not found on parameter \"{}\". Valid targets are {}".format( + dependency_attr, base_dependency, annotations[base_dependency].__annotations__.keys())) + if target_name not in all_deps: + raise ValueError( + "Target dependency \"{}\" was targeted with multiple mapping rules.".format(target_name)) + + # If we've made it this far, the dependency definitely exists. + actual_targets.append(annotations[base_dependency].__annotations__[dependency_attr]) + + all_deps.remove(target_name) + + are_plists = [issubclass(target, ParametrizedList) for target in actual_targets] + if all(are_plists): + # We can do direct sets on everything but ParametrizedLists, so we'll only explicitly + # handle these here. + # TODO(cmikeh2): SPLIT, error if more than 1 + glob_count = src_name.count("*") + if glob_count > 1: + raise ValueError( + "ParametrizedList index inference can only work with a single glob: {}".format(src_name)) + elif glob_count == 0: + raise ValueError( + "Must have wildcard (*) in source name for ParametrizedList mapping: {}".format(src_name)) + + wildcard_idx = src_name.find("*") + prefix = src_name[:wildcard_idx] + suffix = src_name[wildcard_idx + 1:] + attrs[PLIST_HELPERS].append((prefix, suffix, target_or_targets)) + elif any(are_plists): + raise ValueError("Cannot mix ParametrizedLists and Tensors in a single mapping rule.") + + if len(all_deps) > 0: + raise ValueError( + "A parameter mapping was provided for {}, but the following dependencies were not mapped: {}". + format(clsname, all_deps)) + + attrs["finalization_callback"] = make_finalization_callback(all_names) + + new_obj = super().__new__(cls, clsname, bases, attrs) + + setattr(new_obj, "_n_params", len(all_names)) + setattr(new_obj, "_annotation_attrs", all_names) + + return new_obj + + def __call__(cls, *args, **kwargs): + instance = cls.__new__(cls, *args, **kwargs) + instance.__init__(*args, **kwargs) + + for name, annotation in instance.__annotations__.items(): + if issubclass(annotation, ParameterBase): + # TODO(cmikeh2): Do we want to make this a property + # It might also make sense to do this in the base class __init__ + # but since it is tied with the changes made in __new__ it feels + # to me like it should be here. + setattr(instance, name, annotation(instance.inference_model, instance)) + + return instance + + +class LayerContainer(metaclass=LayerMetaclass): # noqa: F811 + """ + Abstract base class for containing model parameters. + + This is primarily a guidance abstraction since we do not put any restrictions + on how the parameters are stored. + + To use this class, annotate the class with `ParameterBase` subclasses and give them + names. As a checkpoint is loaded into this container, the `ParameterBase` instances + will be replaced with realized Tensors as soon as each of their dependencies are met. + + To enable automatic mapping, add a static attribute `PARAM_MAPPING` to the class + definition. This should be a dictionary mapping from a source string to one or + more dependencies. + + ```python + class MyLayer(LayerContainer): + PARAM_MAPPING = { + "path.to.param.dependency", "container_param_1.dependency", + "path.to.param2.dependency", "container_param_2.dependency", + "path.to.param3.*.dependency", "container_param_3.list_dependency" + } + + ... + ``` + """ + + def __init__(self, model: InferenceModel) -> None: + """ + Initialization of the LayerContainer. This method does not need to be overridden + for any children classes. + + Args: + model (InferenceModel): Inference model that will be used to shard and transform + parameters correctly, as well as provide specific information about the model + for `ParameterizedList`s that may be part of one of the member `ParameterBase`s. + """ + self.inference_model = model + self._finalized_params = 0 + + def _initialization_checker(self, check_device: bool = True) -> bool: + """ + Returns whether or not all parameters have been initialized and transformed by + the model. Once this returns True, all the `ParameterBase` instances will be + torch.Tensors. + """ + if self._finalized_params != self.n_params: + return False + + for name in self._annotation_attrs: + tensor = getattr(self, name) + if tensor is None: + continue + elif not isinstance(tensor, InferenceParameter): + raise ValueError("Layer should be finalized, but {} ({}) is neither InferenceParameter or None".format( + name, type(tensor))) + elif check_device and tensor.device != torch.device(get_accelerator().current_device()): + raise RuntimeError("Layer should be finalized, but {} is not on device {}".format( + name, + get_accelerator().current_device())) + return True + + @property + def is_populated(self) -> bool: + """ + Returns whether or not all parameters have been populated by the checkpoint engine, but + does not validat the parameters are on the correct device. + """ + return self._initialization_checker(check_device=False) + + @property + def is_initialized(self) -> bool: + """ + Returns whether or not all parameters have been initialized and transformed by + the model and are located on the appropriate device. Once this returns True, all + the `ParameterBase` instances ``InferenceParameter``s or explicitly set to ``None``. + """ + return self._initialization_checker() + + @property + def n_params(self) -> int: + """ + The number of parameters this container holds. This is a read-only value + that is set by the metaclass. + """ + return self._n_params + + @property + def annotation_attrs(self) -> list: + return self._annotation_attrs + + @property + def mapping_params(self) -> dict: + return getattr(self.__class__, MAPPING_KEY, {}) + + @property + def plist_helpers(self) -> list: + return getattr(self.__class__, PLIST_HELPERS, []) + + def direct_injection(self, name: str, tensor: InferenceParameter) -> None: + + if name not in self._annotation_attrs: + raise ValueError(f"Cannot directly inject {name}, not a valid parameter.") + + setattr(self, name, tensor) + self._finalized_params += 1 + + def set_dependency(self, dep_name: str, dep_value: torch.Tensor) -> None: + """ + Set dependency can be used for managing dependencies when a mapping is provided + in the class definition for the layer. The dep_name here should have any prefix + for transformer layers removed (such as model.layers.*.attn.qkv.weight -> attn.qkv.weight). + + Args: + dep_name (str): The name of the dependency to set. + dep_value (torch.Tensor): The value to set the dependency to. + """ + + def get_dep_name_target(dep_name: str) -> str: + """ + Helper method for getting the target name for a dependency from the + mapping params. Tries to match exact string first, then looks for + wildcards and attempts regex matching. Will return empty string if + no match found. + """ + if dep_name in self.mapping_params: + # If we have an exact match, it's a direct mapping and we can + # immediately set the value. + return self.mapping_params[dep_name] + + matched_targets = [] + for key, target in self.mapping_params.items(): + regex_key = key.replace("*", ".*") + if re.match(regex_key, dep_name): + matched_targets.append(target) + if len(matched_targets) > 1: + raise ValueError(f"Multiple targets matched for dependency {dep_name}: {matched_targets}") + if matched_targets: + return matched_targets[0] + return "" + + if dep_name in self.mapping_params: + # If we have an exact match, it's a direct mapping and we can immediately set + # the value. + target = self.mapping_params[dep_name] + + # Convert single targets to a list for consistency + if isinstance(target, str): + target = [target] + + for target_name in target: + # Double setting doesn't set the attribute correctly, so we do a getattr then setattr + target_param_name, target_dependency_name = target_name.split(".") + target_param = getattr(self, target_param_name) + setattr(target_param, target_dependency_name, dep_value) + return + + # Otherwise we need to map to one of the parameter lists. + for prefix, suffix, dests in self.plist_helpers: + if dep_name.startswith(prefix) and dep_name.endswith(suffix): + # We have a match, so we can set the value. + target_idx = int(dep_name[len(prefix):-len(suffix)]) + + # Convert single targets to a list for consistency + if isinstance(dests, str): + dests = [dests] + + for dest in dests: + target_param_name, target_dependency_name = dest.split(".") + target_param = getattr(self, target_param_name) + target_dependency = getattr(target_param, target_dependency_name) + target_dependency[target_idx] = dep_value + return + + # TODO: Refactor this with the help of cmikeh2 + # We should be able to combine this with the wildcard matching above. + target = get_dep_name_target(dep_name) + if target: + # Convert single targets to a list for consistency + if isinstance(target, str): + target = [target] + + for target_name in target: + # Double setting doesn't set the attribute correctly, so we do a getattr then setattr + target_param_name, target_dependency_name = target_name.split(".") + target_param = getattr(self, target_param_name) + setattr(target_param, target_dependency_name, dep_value) + return + + raise ValueError( + "Could not find a mapping for dependency \"{}\". Check that it is included in the ``MAPPING_PARAMS``. See docstring for more on ``MAPPING_PARAMS``" + .format(dep_name)) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..79605a76a4c28151336040e33e5a87f6ab7ec64b --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .policy import Llama2Policy diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c8bc0b7361e8f371efb3ec0df1d4157902cf009a Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/container.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/container.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c68e4d5e3297638e4d150dbaca36d702f4fc64b9 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/container.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/model.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/model.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..0802e8b2885a46d8460320718befd05b154117d9 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/model.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/policy.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/policy.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..9a863a6a52995e8ee60180716b49aaf99e41dc0b Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/__pycache__/policy.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/container.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/container.py new file mode 100644 index 0000000000000000000000000000000000000000..9de9bdb345743c413841424b3d5ed8755152832a --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/container.py @@ -0,0 +1,78 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# Create a container object to save model-specific tensors using the policy file above. + +from ..common_parameters import * +from ..layer_container_base import LayerContainer +''' + # HF Llama model looks like this: + +LlamaForCausalLM( + (model): LlamaModel( + (embed_tokens): Embedding(32000, 4096, padding_idx=0) + (layers): ModuleList( + (0-31): 32 x LlamaDecoderLayer( + (self_attn): LlamaAttention( + (q_proj): Linear(in_features=4096, out_features=4096, bias=False) + (k_proj): Linear(in_features=4096, out_features=4096, bias=False) + (v_proj): Linear(in_features=4096, out_features=4096, bias=False) + (o_proj): Linear(in_features=4096, out_features=4096, bias=False) + (rotary_emb): LlamaRotaryEmbedding() + ) + (mlp): LlamaMLP( + (gate_proj): Linear(in_features=4096, out_features=11008, bias=False) + (up_proj): Linear(in_features=4096, out_features=11008, bias=False) + (down_proj): Linear(in_features=11008, out_features=4096, bias=False) + (act_fn): SiLUActivation() + ) + (input_layernorm): LlamaRMSNorm() + (post_attention_layernorm): LlamaRMSNorm() + ) + ) + (norm): LlamaRMSNorm() + ) + (lm_head): Linear(in_features=4096, out_features=32000, bias=False) +) +''' + + +class Llama2TransformerContainer(LayerContainer): + """ + Transformer layer container for the Llama-2 model. + """ + qkv_w: UnfusedQKVParameter + attn_out_w: AttentionOutputParameter + mlp_1_w: GatedMLPParameter + mlp_2_w: MLP2Parameter + attn_norm_gamma: NormParameter + mlp_norm_gamma: NormParameter + + PARAM_MAPPING = { + "self_attn.q_proj.weight": "qkv_w.q_params", + "self_attn.k_proj.weight": "qkv_w.k_params", + "self_attn.v_proj.weight": "qkv_w.v_params", + "self_attn.o_proj.weight": "attn_out_w.params", + "mlp.gate_proj.weight": "mlp_1_w.gate_params", + "mlp.up_proj.weight": "mlp_1_w.up_params", + "mlp.down_proj.weight": "mlp_2_w.params", + "input_layernorm.weight": "attn_norm_gamma.params", + "post_attention_layernorm.weight": "mlp_norm_gamma.params", + } + + +class Llama2NonTransformerContainer(LayerContainer): + """ + Non-Transformer layer container for the Llama-2 model. + """ + word_emb: EmbeddingParameter + word_unembed: UnembedParameter + final_norm: NormParameter + + PARAM_MAPPING = { + "model.embed_tokens.weight": "word_emb.params", + "model.norm.weight": "final_norm.params", + "lm_head.weight": "word_unembed.params", + } diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/model.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/model.py new file mode 100644 index 0000000000000000000000000000000000000000..a0c81f4d749e13709e44fd2a10a6912d54f04310 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/model.py @@ -0,0 +1,209 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Optional, Tuple + +import torch + +import deepspeed.comm as dist + +from ...allocator import empty_from +from ...inference_utils import ActivationType, DtypeEnum +from .. import * +from ...modules.configs import * +from ...modules.interfaces import * +from ...ragged import RaggedBatchWrapper + +from .container import Llama2NonTransformerContainer, Llama2TransformerContainer + + +class Llama2InferenceModel(DSTransformerModelBase): + """ + Inference model implementation for ragged batching for Llama-2 models. + """ + + _non_transformer: Optional[Llama2NonTransformerContainer] + """ + Embed + unembed container. Specializing the type annotation. + """ + + _transformer: Optional[Iterable[Llama2TransformerContainer]] + """ + Per-layer transformer container. Specializing the type annotation. + """ + """ + Properties ineherited from `DSInferenceModelBase` + """ + + @property + def max_sequence_length(self) -> int: + return self._config.max_seq_length + + """ + Properties ineherited from `DSTransformerModelBase` + """ + + @property + def num_layers(self) -> int: + return self._config.num_hidden_layers + + @property + def model_dim(self) -> int: + return self._config.hidden_size + + @property + def vocab_size(self) -> int: + return self._config.vocab_size + + @property + def head_size(self) -> int: + return self.model_dim // self.n_heads + + @property + def n_heads(self) -> int: + return self._config.num_attention_heads + + @property + def intermediate_dim(self) -> int: + return self._config.intermediate_size + + @property + def n_heads_kv(self) -> int: + return self._config.num_key_value_heads + + @property + def activation_dtype(self) -> DtypeEnum: + if self._config.torch_dtype == torch.float16: + return DtypeEnum.fp16 + elif self._config.torch_dtype == torch.bfloat16: + return DtypeEnum.bf16 + else: + raise NotImplementedError("Only fp16 and bf16 are supported") + + @property + def mlp_activation_fn(self) -> ActivationType: + activation = self._config.hidden_act.lower() + # llama model family is special and is always gated so force gated versions of relu, gelu, silu + if activation == "gelu": + return ActivationType.GEGLU + elif activation == "relu": + return ActivationType.ReGLU + elif activation == "gegelu": + return ActivationType.GEGLU + elif activation == "silu": + return ActivationType.SiGLU + else: + raise NotImplementedError(f"Activation {activation} not supported") + + @property + def norm_type(self) -> NormTypeEnum: + return NormTypeEnum.RMSNorm + + @property + def positional_embedding_type(self) -> PositionalEmbeddingType: + return PositionalEmbeddingType.rotate_half + + @property + def positional_embedding_config(self) -> Optional[RotateHalfConfig]: + return RotateHalfConfig(theta_base=self._config.rope_theta) + + """ + Forward implementations + """ + + def _forward_embed(self, ragged_batch: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs the embedding lookup prior to running the transformer of the model. + + Arguments: + ragged_batch (RaggedBatchWrapper): The batch to embed. + + Returns: + torch.Tensor: The embedded batch. + """ + embed = self.embed(ragged_batch, self._non_transformer.word_emb) + + if embed.shape[-1] != self.model_dim: + raise ValueError(f"Embedding output shape {embed.shape} does not match model_dim {self.model_dim}") + + return embed + + def _forward_transformer_layer(self, layer_idx: int, residual: torch.Tensor, hidden_states: torch.Tensor, + ragged_batch_info: RaggedBatchWrapper) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Executes one (slightly offset) layer of the transformer. This implementation does a peak-ahead + optimization to fuse the layer norm of the next layer into the current layer. + + Arguments: + layer_idx (int): The index of the layer to execute. + residual (torch.Tensor): The residual tensor from the previous layer. + hidden_states (torch.Tensor): The hidden states from the previous layer. This is the + hidden states after pre normalization. + ragged_batch_info (RaggedBatchWrapper): The batch metadata. + """ + # TODO(cmikeh2): Distribute ragged_batch_info to all modules + + cur_params = self._transformer[layer_idx] + kv_cache = self.state_manager.get_cache(layer_idx) + + hidden_states = self.qkv(hidden_states, cur_params.qkv_w, b=None) + hidden_states = self.attn(hidden_states, kv_cache, ragged_batch_info) + hidden_states = self.attn_out(hidden_states, cur_params.attn_out_w, b=None) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + residual, hidden_states = self.norm(residual, hidden_states, cur_params.mlp_norm_gamma, beta=None) + + # Should be configurable in the future + hidden_states = self.mlp_1(hidden_states, cur_params.mlp_1_w, b=None) + hidden_states = self.mlp_2(hidden_states, cur_params.mlp_2_w, b=None) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + if layer_idx != self.num_layers - 1: + next_params = self._transformer[layer_idx + 1] + residual, hidden_states = self.norm(residual, hidden_states, next_params.attn_norm_gamma, beta=None) + else: + # On last layer, we just need to perform the residual add. Adding into the residual + # here is safe. + residual.add_(hidden_states) + + return residual, hidden_states + + def _forward_unembed(self, hidden_states: torch.Tensor, ragged_batch_info: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs unembedding of the hidden states to logits. This will only sample the final + token of each sequence. + """ + logits = self.unembed(hidden_states, + self._non_transformer.word_unembed, + ragged_batch_info, + gamma=self._non_transformer.final_norm) + + if self.tp_size > 1: + comm_buffer = empty_from(self._comm_logits, (self.tp_size, logits.shape[0], logits.shape[1])) + full_logits = empty_from(self._return_logits, (logits.shape[0], self.vocab_size)) + + dist.all_gather_into_tensor(comm_buffer, logits, group=self._base_mp_group) + + full_logits.copy_(comm_buffer.permute(1, 0, 2).reshape(logits.shape[0], self.vocab_size)) + + return full_logits + else: + return logits + + def forward(self, wrapped_batch: RaggedBatchWrapper) -> torch.Tensor: + + residual = self._forward_embed(wrapped_batch) + + residual, hidden_states = self.norm(residual, None, self._transformer[0].attn_norm_gamma, beta=None) + + for layer_idx in range(self.num_layers): + residual, hidden_states = self._forward_transformer_layer(layer_idx, residual, hidden_states, + wrapped_batch) + + return self._forward_unembed(residual, wrapped_batch) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/policy.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/policy.py new file mode 100644 index 0000000000000000000000000000000000000000..bb13ab6d5bf46fc5f834f69a97d08ca008f7e9c8 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/llama_v2/policy.py @@ -0,0 +1,31 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any + +from ...config_v2 import RaggedInferenceEngineConfig +from ..inference_policy_base import ContainerMap, InferenceV2Policy +from .container import Llama2NonTransformerContainer, Llama2TransformerContainer +from .model import Llama2InferenceModel + + +class Llama2Policy(InferenceV2Policy): + + def instantiate_model(self, engine_config: RaggedInferenceEngineConfig, mp_group: Any) -> Llama2InferenceModel: + return Llama2InferenceModel(config=self._model_config, engine_config=engine_config, base_mp_group=mp_group) + + def build_container_map(self) -> ContainerMap: + map = ContainerMap() + + transformer_containers = [Llama2TransformerContainer(self.model) for _ in range(self.model.num_layers)] + + map.set_transformer_params(['model.layers'], transformer_containers) + + map.set_non_transformer_params(Llama2NonTransformerContainer(self.model)) + + map.set_unmapped_params( + [f'model.layers.{i}.self_attn.rotary_emb.inv_freq' for i in range(self.model.num_layers)]) + + return map diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..60d636693ef3ad9d82af98310ffb7206ec23a781 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .policy import MistralPolicy diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7394a98f2a4ca1517c6bcd9163b92c6c62c60734 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/container.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/container.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c4334fb66c940a5f2ed2fca0b4c8b1fa26fc6196 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/container.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/model.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/model.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e4d8d6479bbfb072c7b2c5db3ef702a9f88768f7 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/model.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/policy.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/policy.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ba1f6b70542180e5d43a7414249443341b0425bb Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/__pycache__/policy.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/container.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/container.py new file mode 100644 index 0000000000000000000000000000000000000000..b4c0956f4049bd1515eb9d95cb865468eb6740b8 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/container.py @@ -0,0 +1,77 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# Create a container object to save model-specific tensors using the policy file above. + +from deepspeed.inference.v2.model_implementations.common_parameters import * +from deepspeed.inference.v2.model_implementations.layer_container_base import LayerContainer +''' + # HF Mistral model (mistralai/Mistral-7B-v0.1) looks like this: +MistralForCausalLM( + (model): MistralModel( + (embed_tokens): Embedding(32000, 4096) + (layers): ModuleList( + (0-31): 32 x MistralDecoderLayer( + (self_attn): MistralAttention( + (q_proj): Linear(in_features=4096, out_features=4096, bias=False) + (k_proj): Linear(in_features=4096, out_features=1024, bias=False) + (v_proj): Linear(in_features=4096, out_features=1024, bias=False) + (o_proj): Linear(in_features=4096, out_features=4096, bias=False) + (rotary_emb): MistralRotaryEmbedding() + ) + (mlp): MistralMLP( + (gate_proj): Linear(in_features=4096, out_features=14336, bias=False) + (up_proj): Linear(in_features=4096, out_features=14336, bias=False) + (down_proj): Linear(in_features=14336, out_features=4096, bias=False) + (act_fn): SiLUActivation() + ) + (input_layernorm): MistralRMSNorm() + (post_attention_layernorm): MistralRMSNorm() + ) + ) + (norm): MistralRMSNorm() + ) + (lm_head): Linear(in_features=4096, out_features=32000, bias=False) +) +''' + + +class MistralTransformerContainer(LayerContainer): + """ + Transformer layer container for the Mistral model. + """ + qkv_w: UnfusedQKVParameter + attn_out_w: AttentionOutputParameter + mlp_1_w: GatedMLPParameter + mlp_2_w: MLP2Parameter + attn_norm_gamma: NormParameter + mlp_norm_gamma: NormParameter + + PARAM_MAPPING = { + "self_attn.q_proj.weight": "qkv_w.q_params", + "self_attn.k_proj.weight": "qkv_w.k_params", + "self_attn.v_proj.weight": "qkv_w.v_params", + "self_attn.o_proj.weight": "attn_out_w.params", + "mlp.gate_proj.weight": "mlp_1_w.gate_params", + "mlp.up_proj.weight": "mlp_1_w.up_params", + "mlp.down_proj.weight": "mlp_2_w.params", + "input_layernorm.weight": "attn_norm_gamma.params", + "post_attention_layernorm.weight": "mlp_norm_gamma.params", + } + + +class MistralNonTransformerContainer(LayerContainer): + """ + Non-Transformer layer container for the Mistral model. + """ + word_emb: EmbeddingParameter + word_unembed: UnembedParameter + final_norm: NormParameter + + PARAM_MAPPING = { + "model.embed_tokens.weight": "word_emb.params", + "model.norm.weight": "final_norm.params", + "lm_head.weight": "word_unembed.params", + } diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/model.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/model.py new file mode 100644 index 0000000000000000000000000000000000000000..318d362f1a64acc117a3e266d88afdd0b336ba56 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/model.py @@ -0,0 +1,207 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Optional, Tuple + +import torch + +import deepspeed.comm as dist + +from ...allocator import empty_from +from ...inference_utils import ActivationType, DtypeEnum +from ...model_implementations import * +from ...modules.configs import * +from ...modules.interfaces import * +from ...ragged import RaggedBatchWrapper + +from .container import MistralNonTransformerContainer, MistralTransformerContainer + + +class MistralInferenceModel(DSTransformerModelBase): + """ + Inference model implementation for ragged batching for Mistral models. + """ + + _non_transformer: Optional[MistralNonTransformerContainer] + """ + Embed + unembed container. Specializing the type annotation. + """ + + _transformer: Optional[Iterable[MistralTransformerContainer]] + """ + Per-layer transformer container. Specializing the type annotation. + """ + """ + Properties ineherited from `DSInferenceModelBase` + """ + + @property + def max_sequence_length(self) -> int: + return self._config.max_seq_length + + """ + Properties ineherited from `DSTransformerModelBase` + """ + + @property + def num_layers(self) -> int: + return self._config.num_hidden_layers + + @property + def model_dim(self) -> int: + return self._config.hidden_size + + @property + def vocab_size(self) -> int: + return self._config.vocab_size + + @property + def head_size(self) -> int: + return self.model_dim // self.n_heads + + @property + def n_heads(self) -> int: + return self._config.num_attention_heads + + @property + def intermediate_dim(self) -> int: + return self._config.intermediate_size + + @property + def n_heads_kv(self) -> int: + return self._config.num_key_value_heads + + @property + def activation_dtype(self) -> DtypeEnum: + if self._config.torch_dtype == torch.float16: + return DtypeEnum.fp16 + elif self._config.torch_dtype == torch.bfloat16: + return DtypeEnum.bf16 + else: + raise NotImplementedError("Only fp16 and bf16 are supported") + + @property + def mlp_activation_fn(self) -> ActivationType: + activation = self._config.hidden_act.lower() + if activation == "gelu": + return ActivationType.GEGLU + elif activation == "relu": + return ActivationType.ReGLU + elif activation == "gegelu": + return ActivationType.GEGLU + elif activation == "silu": + return ActivationType.SiGLU + else: + raise NotImplementedError(f"Activation {activation} not supported") + + @property + def norm_type(self) -> NormTypeEnum: + return NormTypeEnum.RMSNorm + + @property + def positional_embedding_type(self) -> PositionalEmbeddingType: + return PositionalEmbeddingType.rotate_half + + @property + def positional_embedding_config(self) -> Optional[RotateHalfConfig]: + return RotateHalfConfig(theta_base=self._config.rope_theta) + + """ + Forward implementations + """ + + def _forward_embed(self, ragged_batch: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs the embedding lookup prior to running the transformer of the model. + + Arguments: + ragged_batch (RaggedBatchWrapper): The batch to embed. + + Returns: + torch.Tensor: The embedded batch. + """ + embed = self.embed(ragged_batch, self._non_transformer.word_emb) + + if embed.shape[-1] != self.model_dim: + raise ValueError(f"Embedding output shape {embed.shape} does not match model_dim {self.model_dim}") + + return embed + + def _forward_transformer(self, layer_idx: int, residual: torch.Tensor, hidden_states: torch.Tensor, + ragged_batch_info: RaggedBatchWrapper) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Executes one (slightly offset) layer of the transformer. This implementation does a peak-ahead + optimization to fuse the layer norm of the next layer into the current layer. + + Arguments: + layer_idx (int): The index of the layer to execute. + residual (torch.Tensor): The residual tensor from the previous layer. + hidden_states (torch.Tensor): The hidden states from the previous layer. This is the + hidden states after pre normalization. + ragged_batch_info (RaggedBatchWrapper): The batch metadata. + """ + # TODO(cmikeh2): Distribute ragged_batch_info to all modules + + cur_params = self._transformer[layer_idx] + kv_cache = self.state_manager.get_cache(layer_idx) + + hidden_states = self.qkv(hidden_states, cur_params.qkv_w, b=None) + hidden_states = self.attn(hidden_states, kv_cache, ragged_batch_info) + hidden_states = self.attn_out(hidden_states, cur_params.attn_out_w, b=None) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + residual, hidden_states = self.norm(residual, hidden_states, cur_params.mlp_norm_gamma, beta=None) + + # Should be configurable in the future + hidden_states = self.mlp_1(hidden_states, cur_params.mlp_1_w, b=None) + hidden_states = self.mlp_2(hidden_states, cur_params.mlp_2_w, b=None) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + if layer_idx != self.num_layers - 1: + next_params = self._transformer[layer_idx + 1] + residual, hidden_states = self.norm(residual, hidden_states, next_params.attn_norm_gamma, beta=None) + else: + # On last layer, we just need to perform the residual add. Adding into the residual + # here is safe. + residual.add_(hidden_states) + + return residual, hidden_states + + def _forward_unembed(self, hidden_states: torch.Tensor, ragged_batch_info: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs unembedding of the hidden states to logits. This will only sample the final + token of each sequence. + """ + logits = self.unembed(hidden_states, + self._non_transformer.word_unembed, + ragged_batch_info, + gamma=self._non_transformer.final_norm) + + if self.tp_size > 1: + comm_buffer = empty_from(self._comm_logits, (self.tp_size, logits.shape[0], logits.shape[1])) + full_logits = empty_from(self._return_logits, (logits.shape[0], self.vocab_size)) + + dist.all_gather_into_tensor(comm_buffer, logits, group=self._base_mp_group) + + full_logits.copy_(comm_buffer.permute(1, 0, 2).reshape(logits.shape[0], self.vocab_size)) + + return full_logits + else: + return logits + + def forward(self, wrapped_batch: RaggedBatchWrapper) -> torch.Tensor: + + residual = self._forward_embed(wrapped_batch) + + residual, hidden_states = self.norm(residual, None, self._transformer[0].attn_norm_gamma, beta=None) + + for layer_idx in range(self.num_layers): + residual, hidden_states = self._forward_transformer(layer_idx, residual, hidden_states, wrapped_batch) + + return self._forward_unembed(residual, wrapped_batch) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/policy.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/policy.py new file mode 100644 index 0000000000000000000000000000000000000000..b67ec311c952df8f1de538952f4d922cbba6b2ba --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mistral/policy.py @@ -0,0 +1,30 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any + +from ...config_v2 import RaggedInferenceEngineConfig +from ..inference_policy_base import ContainerMap, InferenceV2Policy +from .container import MistralNonTransformerContainer, MistralTransformerContainer +from .model import MistralInferenceModel + + +class MistralPolicy(InferenceV2Policy): + + def instantiate_model(self, engine_config: RaggedInferenceEngineConfig, mp_group: Any) -> MistralInferenceModel: + return MistralInferenceModel(config=self._model_config, engine_config=engine_config, base_mp_group=mp_group) + + def build_container_map(self) -> ContainerMap: + map = ContainerMap() + + transformer_containers = [MistralTransformerContainer(self.model) for _ in range(self.model.num_layers)] + + map.set_transformer_params(['model.layers'], transformer_containers) + + map.set_non_transformer_params(MistralNonTransformerContainer(self.model)) + + map.set_unmapped_params([]) + + return map diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..2cb1aa889291d63dc93bba27494c42e11aa9139f --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .policy import MixtralPolicy diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..62274a0a8d06e31643368d6c4f1ab0d5f2e79571 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/container.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/container.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..1c4fc2188e49f567bc9d84eedb1213561733a835 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/container.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/model.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/model.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..28db73bf51a6b7e7731956c48fdd77d7ec8eed4b Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/model.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/policy.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/policy.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..03a6e9a28c1cab964be522d766aea08d5474a0ae Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/__pycache__/policy.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/container.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/container.py new file mode 100644 index 0000000000000000000000000000000000000000..6ec4a0552b8f53430083e045b913ef5db412f277 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/container.py @@ -0,0 +1,46 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# Create a container object to save model-specific tensors using the policy file above. + +from deepspeed.inference.v2.model_implementations.common_parameters import * +from deepspeed.inference.v2.model_implementations.layer_container_base import LayerContainer + + +class MixtralTransformerContainer(LayerContainer): + + qkv_w: UnfusedQKVParameter + attn_out_w: AttentionOutputParameter + moe_gate: MoEGatingWeightParameter + moe_mlp_1: UnfusedMoEGatedMLPParameter + moe_mlp_2: UnfusedMoEMLP2Parameter + attn_norm_gamma: NormParameter + mlp_norm_gamma: NormParameter + + PARAM_MAPPING = { + "input_layernorm.weight": "attn_norm_gamma.params", + "post_attention_layernorm.weight": "mlp_norm_gamma.params", + "self_attn.q_proj.weight": "qkv_w.q_params", + "self_attn.k_proj.weight": "qkv_w.k_params", + "self_attn.v_proj.weight": "qkv_w.v_params", + "self_attn.o_proj.weight": "attn_out_w.params", + "block_sparse_moe.gate.weight": "moe_gate.params", + "block_sparse_moe.experts.*.w1.weight": "moe_mlp_1.gating_experts", + "block_sparse_moe.experts.*.w3.weight": "moe_mlp_1.up_experts", + "block_sparse_moe.experts.*.w2.weight": "moe_mlp_2.experts", + } + + +class MixtralNonTransformerContainer(LayerContainer): + + word_emb: EmbeddingParameter + word_unembed: UnembedParameter + final_norm: NormParameter + + PARAM_MAPPING = { + "model.embed_tokens.weight": "word_emb.params", + "lm_head.weight": "word_unembed.params", + "model.norm.weight": "final_norm.params", + } diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/model.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/model.py new file mode 100644 index 0000000000000000000000000000000000000000..878cd8e31cec07f3955928e2ac49acdbe97b47b5 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/model.py @@ -0,0 +1,261 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Optional, Tuple + +import torch + +import deepspeed.comm as dist + +from ...allocator import empty_from +from ...config_v2 import RaggedInferenceEngineConfig +from ...inference_utils import ActivationType, DtypeEnum +from ...model_implementations import * +from ...modules.configs import * +from ...modules.interfaces import * +from ...ragged import RaggedBatchWrapper +from ..inference_model_base import ( + DSModelImplementationConfig, + MPType, +) + +from .container import MixtralNonTransformerContainer, MixtralTransformerContainer + + +class MixtralInferenceModel(DSMoETransformerModelBase): + """ + Inference model implementation for Mixtral models. + """ + + _non_transformer: Optional[MixtralNonTransformerContainer] + """ + Embed + unembed container. Specializing the type annotation. + """ + + _transformer: Optional[Iterable[MixtralTransformerContainer]] + """ + Per-layer transformer container. Specializing the type annotation. + """ + """ + Properties ineherited from `DSInferenceModelBase` + """ + + @property + def max_sequence_length(self) -> int: + return self._config.max_position_embeddings + + """ + Properties ineherited from `DSTransformerModelBase` + """ + + @property + def num_layers(self) -> int: + return self._config.num_hidden_layers + + @property + def model_dim(self) -> int: + return self._config.hidden_size + + @property + def vocab_size(self) -> int: + return self._config.vocab_size + + @property + def head_size(self) -> int: + return self.model_dim // self.n_heads + + @property + def n_heads(self) -> int: + return self._config.num_attention_heads + + @property + def intermediate_dim(self) -> int: + return self._config.intermediate_size + + @property + def n_heads_kv(self) -> int: + return self._config.num_key_value_heads + + @property + def activation_dtype(self) -> DtypeEnum: + if self._config.torch_dtype == torch.float16: + return DtypeEnum.fp16 + elif self._config.torch_dtype == torch.bfloat16: + return DtypeEnum.bf16 + else: + raise NotImplementedError("Only fp16 and bf16 are supported") + + @property + def mlp_activation_fn(self) -> ActivationType: + activation = self._config.hidden_act.lower() + if activation == "gelu": + return ActivationType.GEGLU + elif activation == "relu": + return ActivationType.ReGLU + elif activation == "gegelu": + return ActivationType.GEGLU + elif activation == "silu": + return ActivationType.SiGLU + else: + raise NotImplementedError(f"Activation {activation} not supported") + + @property + def norm_type(self) -> NormTypeEnum: + return NormTypeEnum.RMSNorm + + @property + def positional_embedding_type(self) -> PositionalEmbeddingType: + return PositionalEmbeddingType.rotate_half + + @property + def positional_embedding_config(self) -> Optional[RotateHalfConfig]: + """ + The positional embedding configuration for the model. + """ + return RotateHalfConfig(theta_base=self._config.rope_theta) + + """ + Inherited from `DSMoETransformerModelBase` + """ + + @property + def n_experts(self) -> int: + return self._config.num_local_experts + + @property + def n_top_k(self) -> int: + return self._config.num_experts_per_tok + + @property + def normalize_expert_scores(self) -> bool: + return True + + """ + Model implementation + """ + + def __init__(self, config: DSModelImplementationConfig, engine_config: RaggedInferenceEngineConfig, + base_mp_group: MPType) -> None: + """ + Base implementation for initialization. By default, this will initialize + the traditional components of a transformer model: + - Embedding + - QKV projection + - Self attention + - Attention output projection + - Feed forward network + - Normalization + - Unembedding + + Arguments: + config (DSModelImplementationConfig): Model-specific configuration. No assumptions + should be made about this config that are not closely tied to the specific + model implementation. + engine_config (RaggedInferenceEngineConfig): Engine configuration. + base_mp_group (MPType): Base communication group for Tensor-parallel inference. + """ + super().__init__(config, engine_config, base_mp_group) + + self.make_norm_layer() + self.make_qkv_layer() + self.make_attn_layer() + self.make_attn_out_layer() + self.make_moe_layer() + self.make_embedding_layer() + self.make_unembedding_layer() + self._kv_cache_config = None + + def _forward_embed(self, ragged_batch: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs the embedding lookup prior to running the transformer of the model. + + Arguments: + ragged_batch (RaggedBatchWrapper): The batch to embed. + + Returns: + torch.Tensor: The embedded batch. + """ + embed = self.embed(ragged_batch, self._non_transformer.word_emb) + + if embed.shape[-1] != self.model_dim: + raise ValueError(f"Embedding output shape {embed.shape} does not match model_dim {self.model_dim}") + + return embed + + def _forward_transformer(self, layer_idx: int, residual: torch.Tensor, hidden_states: torch.Tensor, + ragged_batch_info: RaggedBatchWrapper) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Executes one (slightly offset) layer of the transformer. This implementation does a peak-ahead + optimization to fuse the layer norm of the next layer into the current layer. + + Arguments: + layer_idx (int): The index of the layer to execute. + residual (torch.Tensor): The residual tensor from the previous layer. + hidden_states (torch.Tensor): The hidden states from the previous layer. This is the + hidden states after pre normalization. + ragged_batch_info (RaggedBatchWrapper): The batch metadata. + """ + # TODO(cmikeh2): Distribute ragged_batch_info to all modules + + cur_params = self._transformer[layer_idx] + kv_cache = self.state_manager.get_cache(layer_idx) + + hidden_states = self.qkv(hidden_states, cur_params.qkv_w) + hidden_states = self.attn(hidden_states, kv_cache, ragged_batch_info) + hidden_states = self.attn_out(hidden_states, cur_params.attn_out_w) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + residual, hidden_states = self.norm(residual, hidden_states, cur_params.mlp_norm_gamma) + + hidden_states = self.moe(hidden_states, ragged_batch_info, cur_params.moe_gate, cur_params.moe_mlp_1, + cur_params.moe_mlp_2) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + if layer_idx != self.num_layers - 1: + next_params = self._transformer[layer_idx + 1] + residual, hidden_states = self.norm(residual, hidden_states, next_params.attn_norm_gamma) + else: + # On last layer, we just need to perform the residual add. Adding into the residual + # here is safe. + residual.add_(hidden_states) + + return residual, hidden_states + + def _forward_unembed(self, hidden_states: torch.Tensor, ragged_batch_info: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs unembedding of the hidden states to logits. This will only sample the final + token of each sequence. + """ + logits = self.unembed(hidden_states, + self._non_transformer.word_unembed, + ragged_batch_info, + gamma=self._non_transformer.final_norm) + + if self.tp_size > 1: + comm_buffer = empty_from(self._comm_logits, (self.tp_size, logits.shape[0], logits.shape[1])) + full_logits = empty_from(self._return_logits, (logits.shape[0], self.vocab_size)) + + dist.all_gather_into_tensor(comm_buffer, logits, group=self._base_mp_group) + + full_logits.copy_(comm_buffer.permute(1, 0, 2).reshape(logits.shape[0], self.vocab_size)) + + return full_logits + else: + return logits + + def forward(self, wrapped_batch: RaggedBatchWrapper) -> torch.Tensor: + + residual = self._forward_embed(wrapped_batch) + + residual, hidden_states = self.norm(residual, None, self._transformer[0].attn_norm_gamma, beta=None) + + for layer_idx in range(self.num_layers): + residual, hidden_states = self._forward_transformer(layer_idx, residual, hidden_states, wrapped_batch) + + return self._forward_unembed(residual, wrapped_batch) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/policy.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/policy.py new file mode 100644 index 0000000000000000000000000000000000000000..2f0087919720d040ee53eda7723268f2635db585 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/mixtral/policy.py @@ -0,0 +1,31 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any + +from ...config_v2 import RaggedInferenceEngineConfig +from ..inference_policy_base import ContainerMap, InferenceV2Policy +from .container import MixtralTransformerContainer, MixtralNonTransformerContainer +from .model import MixtralInferenceModel + + +class MixtralPolicy(InferenceV2Policy): + + def instantiate_model(self, engine_config: RaggedInferenceEngineConfig, mp_group: Any) -> MixtralInferenceModel: + return MixtralInferenceModel(config=self._model_config, engine_config=engine_config, base_mp_group=mp_group) + + def build_container_map(self) -> ContainerMap: + + map = ContainerMap() + + transformer_containers = [MixtralTransformerContainer(self.model) for _ in range(self.model.num_layers)] + + map.set_transformer_params(['model.layers'], transformer_containers) + + map.set_non_transformer_params(MixtralNonTransformerContainer(self.model)) + + map.set_unmapped_params([]) + + return map diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..c0f24d5243b820748a9ddb9a1380fd51644ae917 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .policy import OPTPolicy diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..0d325eb6a508252580e5d56dfbb7221e308389e4 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/container.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/container.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..13bbb8e42f9af58d465316c99b9b5103226d4a2e Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/container.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/model.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/model.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..dfdf572d303818c245610ee63a390a7833fd272e Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/model.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/policy.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/policy.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..1e1faef2ee7114a179e51ef9ee1f830c97376dd1 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/__pycache__/policy.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/container.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/container.py new file mode 100644 index 0000000000000000000000000000000000000000..e97599ef8e50e3dc30ce5540bd4721bb055df87e --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/container.py @@ -0,0 +1,94 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# Create a container object to save model-specific tensors using the policy file above. + +from ..common_parameters import * +from ..layer_container_base import LayerContainer +''' + # HF OPT model looks like this: + +OPTForCausalLM( + (model): OPTModel( + (decoder): OPTDecoder( + (embed_tokens): Embedding(50272, 768, padding_idx=1) + (embed_positions): OPTLearnedPositionalEmbedding(2050, 768) + (final_layer_norm): LayerNorm((768,), eps=1e-05, elementwise_affine=True) + (layers): ModuleList( + (0-11): 12 x OPTDecoderLayer( + (self_attn): OPTAttention( + (k_proj): Linear(in_features=768, out_features=768, bias=True) + (v_proj): Linear(in_features=768, out_features=768, bias=True) + (q_proj): Linear(in_features=768, out_features=768, bias=True) + (out_proj): Linear(in_features=768, out_features=768, bias=True) + ) + (activation_fn): ReLU() + (self_attn_layer_norm): LayerNorm((768,), eps=1e-05, elementwise_affine=True) + (fc1): Linear(in_features=768, out_features=3072, bias=True) + (fc2): Linear(in_features=3072, out_features=768, bias=True) + (final_layer_norm): LayerNorm((768,), eps=1e-05, elementwise_affine=True) + ) + ) + ) + ) + (lm_head): Linear(in_features=768, out_features=50272, bias=False) +) + +''' + + +class OPTTransformerContainer(LayerContainer): + """ + Transformer layer container for the OPT model. + """ + qkv_w: UnfusedQKVParameter + qkv_b: UnfusedQKVParameter + attn_out_w: AttentionOutputParameter + attn_out_b: AttentionOutputParameter + mlp_1_w: MLP1Parameter + mlp_1_b: MLP1Parameter + mlp_2_w: MLP2Parameter + mlp_2_b: MLP2Parameter + attn_norm_beta: NormParameter + attn_norm_gamma: NormParameter + mlp_norm_beta: NormParameter + mlp_norm_gamma: NormParameter + + PARAM_MAPPING = { + "self_attn.q_proj.weight": "qkv_w.q_params", + "self_attn.q_proj.bias": "qkv_b.q_params", + "self_attn.k_proj.weight": "qkv_w.k_params", + "self_attn.k_proj.bias": "qkv_b.k_params", + "self_attn.v_proj.weight": "qkv_w.v_params", + "self_attn.v_proj.bias": "qkv_b.v_params", + "self_attn.out_proj.weight": "attn_out_w.params", + "self_attn.out_proj.bias": "attn_out_b.params", + "fc1.weight": "mlp_1_w.params", + "fc1.bias": "mlp_1_b.params", + "fc2.weight": "mlp_2_w.params", + "fc2.bias": "mlp_2_b.params", + "self_attn_layer_norm.weight": "attn_norm_gamma.params", + "self_attn_layer_norm.bias": "attn_norm_beta.params", + "final_layer_norm.weight": "mlp_norm_gamma.params", + "final_layer_norm.bias": "mlp_norm_beta.params", + } + + +class OPTNonTransformerContainer(LayerContainer): + """ + Non-Transformer layer container for the OPT model. + """ + word_emb: EmbeddingParameter + word_emb_pos: EmbeddingParameter + word_unembed: UnembedParameter + final_norm_w: NormParameter + final_norm_b: NormParameter + + PARAM_MAPPING = { + "*decoder.embed_tokens.weight": ["word_emb.params", "word_unembed.params"], + "*decoder.embed_positions.weight": "word_emb_pos.params", + "*decoder.final_layer_norm.weight": "final_norm_w.params", + "*decoder.final_layer_norm.bias": "final_norm_b.params", + } diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/model.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/model.py new file mode 100644 index 0000000000000000000000000000000000000000..adf011d8f1a7884f47569ee186336dcc77b355bb --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/model.py @@ -0,0 +1,197 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Optional, Tuple + +import torch + +import deepspeed.comm as dist + +from ...allocator import empty_from +from ...inference_utils import ActivationType, DtypeEnum +from ...model_implementations import * +from ...modules.configs import * +from ...ragged import RaggedBatchWrapper +from .container import OPTNonTransformerContainer, OPTTransformerContainer + +from ...modules.heuristics import instantiate_embed + + +class OPTInferenceModel(DSTransformerModelBase): + """ + Inference model implementation for ragged batching for OPT models. + """ + + _non_transformer: Optional[OPTNonTransformerContainer] + """ + Embed + unembed container. Specializing the type annotation. + """ + + _transformer: Optional[Iterable[OPTTransformerContainer]] + """ + Per-layer transformer container. Specializing the type annotation. + """ + """ + Properties ineherited from `DSInferenceModelBase` + """ + + @property + def max_sequence_length(self) -> int: + return self._config.max_seq_length + + """ + Properties ineherited from `DSTransformerModelBase` + """ + + @property + def num_layers(self) -> int: + return self._config.num_hidden_layers + + @property + def model_dim(self) -> int: + return self._config.hidden_size + + @property + def vocab_size(self) -> int: + return self._config.vocab_size + + @property + def head_size(self) -> int: + return self.model_dim // self.n_heads + + @property + def n_heads(self) -> int: + return self._config.num_attention_heads + + @property + def intermediate_dim(self) -> int: + return self._config.ffn_dim + + @property + def activation_dtype(self) -> DtypeEnum: + if self._config.torch_dtype == torch.float16: + return DtypeEnum.fp16 + elif self._config.torch_dtype == torch.bfloat16: + return DtypeEnum.bf16 + else: + raise NotImplementedError("Only fp16 and bf16 are supported") + + @property + def mlp_activation_fn(self) -> ActivationType: + return ActivationType.RELU + + @property + def norm_type(self) -> NormTypeEnum: + return NormTypeEnum.LayerNorm + + @property + def positional_embedding_type(self) -> PositionalEmbeddingType: + return PositionalEmbeddingType.none + + @property + def positional_embedding_config(self) -> Optional[RotateHalfConfig]: + return None + + """ + Overrides of ``DSTransformerModelBase`` methods + """ + + def make_embedding_layer(self) -> None: + """ + Performs setup and creates embedding DSModule. Since OPT includes trained + positional embeddings, we will override the base model implementation. + """ + + embed_config = DSEmbeddingsConfig(max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + residual_dtype=self.activation_dtype, + embedding_dim=self.model_dim, + positional_embedding=True, + positional_offset=2) + + self.embed = instantiate_embed(embed_config, self._engine_config) + + """ + Forward implementations + """ + + def _forward_embed(self, ragged_batch: RaggedBatchWrapper) -> torch.Tensor: + embed = self.embed(ragged_batch, self._non_transformer.word_emb, self._non_transformer.word_emb_pos) + if embed.shape[-1] != self.model_dim: + raise ValueError(f"Embedding output shape {embed.shape} does not match model_dim {self.model_dim}") + + return embed + + def _forward_transformer_layer(self, layer_idx: int, residual: torch.Tensor, hidden_states: torch.Tensor, + ragged_batch_info: RaggedBatchWrapper) -> Tuple[torch.Tensor, torch.Tensor]: + # TODO(cmikeh2): Distribute ragged_batch_info to all modules + + cur_params = self._transformer[layer_idx] + kv_cache = self.state_manager.get_cache(layer_idx) + + hidden_states = self.qkv(hidden_states, cur_params.qkv_w, b=cur_params.qkv_b) + hidden_states = self.attn(hidden_states, kv_cache, ragged_batch_info) + hidden_states = self.attn_out(hidden_states, cur_params.attn_out_w, b=cur_params.attn_out_b) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + residual, hidden_states = self.norm(residual, + hidden_states, + cur_params.mlp_norm_gamma, + beta=cur_params.mlp_norm_beta) + + # Should be configurable in the future + hidden_states = self.mlp_1(hidden_states, cur_params.mlp_1_w, b=cur_params.mlp_1_b) + hidden_states = self.mlp_2(hidden_states, cur_params.mlp_2_w, b=cur_params.mlp_2_b) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + if layer_idx != self.num_layers - 1: + next_params = self._transformer[layer_idx + 1] + residual, hidden_states = self.norm(residual, + hidden_states, + next_params.attn_norm_gamma, + beta=next_params.attn_norm_beta) + else: + # On last layer, we just need to perform the residual add. Adding into the residual + # here is safe. + residual.add_(hidden_states) + + return residual, hidden_states + + def _forward_unembed(self, hidden_states: torch.Tensor, ragged_batch_info: RaggedBatchWrapper) -> torch.Tensor: + logits = self.unembed(hidden_states, + self._non_transformer.word_unembed, + ragged_batch_info, + gamma=self._non_transformer.final_norm_w, + beta=self._non_transformer.final_norm_b) + + if self.tp_size > 1: + comm_buffer = empty_from(self._comm_logits, (self.tp_size, logits.shape[0], logits.shape[1])) + full_logits = empty_from(self._return_logits, (logits.shape[0], self.vocab_size)) + + dist.all_gather_into_tensor(comm_buffer, logits, group=self._base_mp_group) + + full_logits.copy_(comm_buffer.permute(1, 0, 2).reshape(logits.shape[0], self.vocab_size)) + + return full_logits + else: + return logits + + def forward(self, wrapped_batch: RaggedBatchWrapper) -> torch.Tensor: + + residual = self._forward_embed(wrapped_batch) + + residual, hidden_states = self.norm(residual, + None, + self._transformer[0].attn_norm_gamma, + beta=self._transformer[0].attn_norm_beta) + + for layer_idx in range(self.num_layers): + residual, hidden_states = self._forward_transformer_layer(layer_idx, residual, hidden_states, + wrapped_batch) + + return self._forward_unembed(residual, wrapped_batch) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/policy.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/policy.py new file mode 100644 index 0000000000000000000000000000000000000000..d57d5beb48d515fbd6d4c5a0741edf1a4867136f --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/opt/policy.py @@ -0,0 +1,30 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any + +from ...config_v2 import RaggedInferenceEngineConfig +from ..inference_policy_base import ContainerMap, InferenceV2Policy +from .container import OPTNonTransformerContainer, OPTTransformerContainer +from .model import OPTInferenceModel + + +class OPTPolicy(InferenceV2Policy): + + def instantiate_model(self, engine_config: RaggedInferenceEngineConfig, mp_group: Any) -> OPTInferenceModel: + return OPTInferenceModel(config=self._model_config, engine_config=engine_config, base_mp_group=mp_group) + + def build_container_map(self) -> ContainerMap: + map = ContainerMap() + + transformer_containers = [OPTTransformerContainer(self.model) for _ in range(self.model.num_layers)] + + map.set_transformer_params(['model.decoder.layers', 'decoder.layers'], transformer_containers) + + map.set_non_transformer_params(OPTNonTransformerContainer(self.model)) + + map.set_unmapped_params(['lm_head.weight']) + + return map diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/parameter_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/parameter_base.py new file mode 100644 index 0000000000000000000000000000000000000000..2dcb63c050a0b4ceea8ce7ece38b6fcd329273a8 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/parameter_base.py @@ -0,0 +1,255 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import weakref +from abc import abstractmethod +from typing import Type + +import torch + +# Currently have dependency loops for the type hints. +InferenceModel = Type["InferenceModel"] +LayerContainer = Type["LayerContainer"] + +MAPPING_KEY = "PARAM_MAPPING" + + +def make_param_getter(clsname, param): + """ + Normal getter implementation for a property. + """ + + def param_getter(self): + return getattr(self, f"__{clsname}__{param}") + + return param_getter + + +def make_param_setter(clsname, param): + """ + Setter implementation that will call complete component to potentially + finalize the parameter. + """ + + def param_setter(self, value): + setattr(self, f"__{clsname}__{param}", value) + self.complete_component() + + return param_setter + + +def make_readonly_setter(): + """ + Setter implementation that will raise an error if called. + """ + + def paramlist_setter(self, value): + raise ValueError("Cannot set a ParametrizedList directly.") + + return paramlist_setter + + +class ParameterMetaclass(type): + """ + MetaClass for the ParameterBase base class. This class will parse the `src_params` + attribute and create properties for each of the dependencies. A dependency can either + be represented as a string, which is interpreted as a named Tensor, or a `ParametrizedList` + subclass. + """ + + def __new__(cls, clsname, bases, attrs): + + annotations = attrs.get("__annotations__", {}) + dependencies = { + name: annotation + for name, annotation in annotations.items() if issubclass(annotation, (torch.Tensor, ParametrizedList)) + } + n_dependencies = len(dependencies) + + # Create properties for each of our dependencies + for d_name, d_type in dependencies.items(): + if issubclass(d_type, ParametrizedList): + assert hasattr( + d_type, "count_attr" + ), "ParametrizedList must have a count_attr attribute to access on the inference module." + attrs[d_name] = property(make_param_getter(clsname, d_name), make_readonly_setter()) + else: # torch.Tensor + attrs[d_name] = property(make_param_getter(clsname, d_name), make_param_setter(clsname, d_name)) + + new_cls = super().__new__(cls, clsname, bases, attrs) + new_cls.n_dependencies = n_dependencies + + return new_cls + + def __call__(cls, *args, **kwargs): + new_obj = super().__call__(*args, **kwargs) + new_obj.__init__(*args, **kwargs) + + setattr(new_obj, "dest_param", None) + + # Initialize our dependences to None/empty `ParametrizedList`s + for name, annotation in new_obj.__annotations__.items(): + if issubclass(annotation, ParametrizedList): + #TODO(jeff): update assert with this, model implementation attribute does not align or missing wrt the ParametrizedList attributes + assert hasattr( + new_obj.inference_model, annotation.count_attr + ), f"new_obj={new_obj.__class__.__name__}, name={name}, annotation.count_attr={annotation.count_attr}" + param_list = annotation(new_obj, getattr(new_obj.inference_model, annotation.count_attr)) + setattr(new_obj, f"__{new_obj.__class__.__name__}__{name}", param_list) + else: # torch.Tensor + setattr(new_obj, f"__{new_obj.__class__.__name__}__{name}", None) + + return new_obj + + +class ParameterBase(metaclass=ParameterMetaclass): + """ + A ParameterBase allows us to consolidate tracking the dependencies of loading a parameter from + a checkpoint into a single object. This class should not be used directly, but rather subclassed + and the `src_params` attribute set to a list of strings and/or `ParametrizedList`s. + """ + + # inference_model: InferenceModel + """ + Inference model that will provide context on how to shard and transform the parameter. + """ + + #completed_components: int + """ + How many of the layer dependencies have been met. This is used to determine when the parameter + is ready to be finalized. A ParametrizedList counts as a single dependency for the purposes + of this counter. + """ + + def __init__(self, model: InferenceModel, parent_container: LayerContainer) -> None: + """ + Direct constructor. This should not be called from client code. + + Args: + model (InferenceModel): Inference model that will be used to shard and transform the + parameter in `finalize`. + parent_container (LayerContainer): The parent container that this parameter is a member + of. We will build a weakref to this container to call the finalization callback. + """ + self.inference_model = model + self.completed_components = 0 + self.parent_container = weakref.ref(parent_container) + + @abstractmethod + def finalize(self) -> torch.Tensor: + """ + Finalize the parameter after all of its source parameters have been set. This method + will be automatically called when all inputs have been set. It should return the Tensor + with all transformations performed on it. + """ + pass + + def complete_component(self) -> None: + """ + Mark a component as completed. This should be called by the relevant setter of a direct + property or a ParametrizedList. This method will automatically call `finalize` when all + dependencies have been met and then call the finalization callback on the parent container. + + Once the finalization callback has been called, the parameter will be replaced with the + `dst_param` attribute on the parent container, and this instance will be destroyed. + """ + self.completed_components += 1 + + if self.completed_components != self.n_dependencies: + return + + finalized_param = self.finalize() + self.parent_container().finalization_callback(self, finalized_param) + + +class ParametrizedList: + """ + A ParametrizedList is a list of parameters that are dependencies + of a `ParameterBase` but may vary in length depending on the model + configuration (rather than architecture). For example, a MoE layer + may have different number of experts depending on the size of the model. + + This class is used to manage these lists and provide integer indexing + of a single component rather than accessing names directly. For example, + it tends to be more natural to access the 8th expert with `experts[8]` + rather than a name like `expert_8`, especially as an attribute. + + To inherit from this class, set static variables `name` and `count_attr`. + + ```python + class MyParametrizedList(ParametrizedList): + count_attr: str = "my_list_count" + ``` + + In the above example, `my_list_count` should be an accessible attribute + of the inference model (i.e. via `self.inference_model.my_list_count`). + + NOTE: There are some APIs in which this type cannot be used as if it is + just a list of Tensors. For example, `torch.cat(param_list)` will not work. + However, you can make it compatible with a tuple wrapper: + `torch.cat(tuple(param_list))` + """ + + n_params: int + """ + Number of params this list contains. + """ + + param: ParameterBase + """ + WeakRef to the owning parameter. + """ + + def __init__(self, param: ParameterBase, n_params: int) -> None: + """ + Constructor. Should not be called from client code. + + Args: + param (ParameterBase): The owning parameter. + n_params (int): The number of parameters this list contains. This should be + """ + self.n_params = n_params + self.set_params = 0 + self.param = weakref.ref(param) + self._params = [None] * n_params + + def __getitem__(self, index): + return self._params[index] + + def __setitem__(self, index, value): + if self._params[index] is not None: + raise ValueError("Cannot set a parameter twice.") + + self._params[index] = value + self.set_params += 1 + + if self.set_params != self.n_params: + return + + self.param().complete_component() + + def __iter__(self): + return iter(self._params) + + +def ParamList(attr: str): + """ + Helper to create a subclass of ParametrizedList with the desired `count_attr`. + + In this manner, we can annotate the type of a Parameter dependency with the + following: + + ```python + class CustomParameter(ParameterBase): + dependency_list: ParamList("dependencies_count_name") + ``` + + where "dependencies_count_name" is the name of the attribute on the inference model. + """ + + class ParametrizedListInstance(ParametrizedList): + count_attr: str = attr + + return ParametrizedListInstance diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..3ab107e75a9147f30176f9e3f6bc575898d8e572 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .policy import PhiPolicy diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..234d114e06cc266fe679f7d17b6e02b9289d4cd3 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/containers.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/containers.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b4b8112e7cb80cc2ac5582ca1825080a0984eaeb Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/containers.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/model.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/model.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e88e13851c5c330d52e7290c79814a1c41230a71 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/model.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/policy.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/policy.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..cefa1600126e5c7e01c5c2a9e2562a8a9a4b94cc Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/__pycache__/policy.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/containers.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/containers.py new file mode 100644 index 0000000000000000000000000000000000000000..21f07eb8c99a037243086688b551bbe3fa9ec3d6 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/containers.py @@ -0,0 +1,91 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# Create a container object to save model-specific tensors using the policy file above. + +from ..common_parameters import * +from ..layer_container_base import LayerContainer +''' + # HF Phi-2 model looks like this: + +PhiForCausalLM( + (model): PhiModel( + (embed_tokens): Embedding(51200, 2560) + (embed_dropout): Dropout(p=0.0, inplace=False) + (layers): ModuleList( + (0-31): 32 x PhiDecoderLayer( + (self_attn): PhiAttention( + (q_proj): Linear(in_features=2560, out_features=2560, bias=True) + (k_proj): Linear(in_features=2560, out_features=2560, bias=True) + (v_proj): Linear(in_features=2560, out_features=2560, bias=True) + (dense): Linear(in_features=2560, out_features=2560, bias=True) + (rotary_emb): PhiRotaryEmbedding() + ) + (mlp): PhiMLP( + (activation_fn): NewGELUActivation() + (fc1): Linear(in_features=2560, out_features=10240, bias=True) + (fc2): Linear(in_features=10240, out_features=2560, bias=True) + ) + (input_layernorm): LayerNorm((2560,), eps=1e-05, elementwise_affine=True) + (resid_dropout): Dropout(p=0.1, inplace=False) + ) + ) + (final_layernorm): LayerNorm((2560,), eps=1e-05, elementwise_affine=True) + ) + (lm_head): Linear(in_features=2560, out_features=51200, bias=True) +) +''' + + +class PhiTransformerContainer(LayerContainer): + """ + Transformer layer container for the Phi model. + """ + qkv_w: UnfusedQKVParameter + qkv_b: UnfusedQKVParameter + attn_out_w: AttentionOutputParameter + attn_out_b: AttentionOutputParameter + mlp_1_w: MLP1Parameter + mlp_1_b: MLP1Parameter + mlp_2_w: MLP2Parameter + mlp_2_b: MLP2Parameter + ln_gamma: NormParameter + ln_beta: NormParameter + + PARAM_MAPPING = { + "self_attn.q_proj.weight": "qkv_w.q_params", + "self_attn.k_proj.weight": "qkv_w.k_params", + "self_attn.v_proj.weight": "qkv_w.v_params", + "self_attn.q_proj.bias": "qkv_b.q_params", + "self_attn.k_proj.bias": "qkv_b.k_params", + "self_attn.v_proj.bias": "qkv_b.v_params", + "self_attn.dense.weight": "attn_out_w.params", + "self_attn.dense.bias": "attn_out_b.params", + "mlp.fc1.weight": "mlp_1_w.params", + "mlp.fc1.bias": "mlp_1_b.params", + "mlp.fc2.weight": "mlp_2_w.params", + "mlp.fc2.bias": "mlp_2_b.params", + "input_layernorm.weight": "ln_gamma.params", + "input_layernorm.bias": "ln_beta.params", + } + + +class PhiNonTransformerContainer(LayerContainer): + """ + Non-Transformer layer container for the Phi model. + """ + word_emb: EmbeddingParameter + word_unembed_w: UnembedParameter + word_unembed_b: UnembedParameter + final_norm_gamma: NormParameter + final_norm_beta: NormParameter + + PARAM_MAPPING = { + "model.embed_tokens.weight": "word_emb.params", + "model.final_layernorm.weight": "final_norm_gamma.params", + "model.final_layernorm.bias": "final_norm_beta.params", + "lm_head.weight": "word_unembed_w.params", + "lm_head.bias": "word_unembed_b.params", + } diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/model.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/model.py new file mode 100644 index 0000000000000000000000000000000000000000..2d5826810cb57bfb7fe9b7df75edf1cf947bd1f8 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/model.py @@ -0,0 +1,199 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Optional, Tuple + +import torch + +import deepspeed.comm as dist + +from ...allocator import empty_from +from ...inference_utils import ActivationType, DtypeEnum +from .. import * +from ...modules.configs import * +from ...modules.interfaces import * +from ...ragged import RaggedBatchWrapper + +from .containers import PhiNonTransformerContainer, PhiTransformerContainer + + +class PhiInferenceModel(DSTransformerModelBase): + """ + Inference model implementation for ragged batching for Llama-2 models. + """ + + _non_transformer: Optional[PhiNonTransformerContainer] + """ + Embed + unembed container. Specializing the type annotation. + """ + + _transformer: Optional[Iterable[PhiTransformerContainer]] + """ + Per-layer transformer container. Specializing the type annotation. + """ + """ + Properties inherited from `DSInferenceModelBase` + """ + + @property + def max_sequence_length(self) -> int: + return self._config.max_seq_length + + """ + Properties inherited from `DSTransformerModelBase` + """ + + @property + def num_layers(self) -> int: + return self._config.num_hidden_layers + + @property + def model_dim(self) -> int: + return self._config.hidden_size + + @property + def vocab_size(self) -> int: + return self._config.vocab_size + + @property + def head_size(self) -> int: + return self.model_dim // self.n_heads + + @property + def n_heads(self) -> int: + return self._config.num_attention_heads + + @property + def intermediate_dim(self) -> int: + return self._config.intermediate_size + + @property + def n_heads_kv(self) -> int: + return self._config.num_key_value_heads + + @property + def activation_dtype(self) -> DtypeEnum: + if self._config.torch_dtype == torch.float16: + return DtypeEnum.fp16 + elif self._config.torch_dtype == torch.bfloat16: + return DtypeEnum.bf16 + else: + raise NotImplementedError("Only fp16 and bf16 are supported") + + @property + def mlp_activation_fn(self) -> ActivationType: + return ActivationType.GELU + + @property + def norm_type(self) -> NormTypeEnum: + return NormTypeEnum.LayerNorm + + @property + def positional_embedding_type(self) -> PositionalEmbeddingType: + return PositionalEmbeddingType.rotate_half + + @property + def positional_embedding_config(self) -> Optional[RotateHalfConfig]: + rotary_dim = int(self._config.partial_rotary_factor * self.head_size) + return RotateHalfConfig(rotate_dim=rotary_dim, theta_base=self._config.rope_theta) + + """ + Forward implementations + """ + + def _forward_embed(self, ragged_batch: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs the embedding lookup prior to running the transformer of the model. + + Arguments: + ragged_batch (RaggedBatchWrapper): The batch to embed. + + Returns: + torch.Tensor: The embedded batch. + """ + embed = self.embed(ragged_batch, self._non_transformer.word_emb) + + if embed.shape[-1] != self.model_dim: + raise ValueError(f"Embedding output shape {embed.shape} does not match model_dim {self.model_dim}") + + return embed + + def _forward_transformer_layer(self, layer_idx: int, residual: torch.Tensor, hidden_states: torch.Tensor, + ragged_batch_info: RaggedBatchWrapper) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Executes one (slightly offset) layer of the transformer. This implementation does a peak-ahead + optimization to fuse the layer norm of the next layer into the current layer. + + Arguments: + layer_idx (int): The index of the layer to execute. + residual (torch.Tensor): The residual tensor from the previous layer. + hidden_states (torch.Tensor): The hidden states from the previous layer. This is the + hidden states after pre normalization. + ragged_batch_info (RaggedBatchWrapper): The batch metadata. + """ + cur_params = self._transformer[layer_idx] + kv_cache = self.state_manager.get_cache(layer_idx) + + attn_ln_out = hidden_states + attn_hidden_state = self.qkv(attn_ln_out, cur_params.qkv_w, b=cur_params.qkv_b) + attn_hidden_state = self.attn(attn_hidden_state, kv_cache, ragged_batch_info) + attention_output = self.attn_out(attn_hidden_state, cur_params.attn_out_w, b=cur_params.attn_out_b) + + mlp_ln_out = hidden_states + mlp_hidden_state = self.mlp_1(mlp_ln_out, cur_params.mlp_1_w, b=cur_params.mlp_1_b) + mlp_output = self.mlp_2(mlp_hidden_state, cur_params.mlp_2_w, b=cur_params.mlp_2_b) + + mlp_output.add_(attention_output) + + if self.tp_size > 1: + dist.all_reduce(mlp_output, group=self._base_mp_group) + + if layer_idx != self.num_layers - 1: + next_params = self._transformer[layer_idx + 1] + residual, mlp_output = self.norm(residual, mlp_output, next_params.ln_gamma, beta=next_params.ln_beta) + else: + # On last layer, we just need to perform the residual add. Adding into the residual + # here is safe. + residual.add_(mlp_output) + + return residual, mlp_output + + def _forward_unembed(self, hidden_states: torch.Tensor, ragged_batch_info: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs unembedding of the hidden states to logits. This will only sample the final + token of each sequence. + """ + logits = self.unembed(hidden_states, + self._non_transformer.word_unembed_w, + ragged_batch_info, + bias=self._non_transformer.word_unembed_b, + gamma=self._non_transformer.final_norm_gamma, + beta=self._non_transformer.final_norm_beta) + + if self.tp_size > 1: + comm_buffer = empty_from(self._comm_logits, (self.tp_size, logits.shape[0], logits.shape[1])) + full_logits = empty_from(self._return_logits, (logits.shape[0], self.vocab_size)) + + dist.all_gather_into_tensor(comm_buffer, logits, group=self._base_mp_group) + + full_logits.copy_(comm_buffer.permute(1, 0, 2).reshape(logits.shape[0], self.vocab_size)) + + return full_logits + else: + return logits + + def forward(self, wrapped_batch: RaggedBatchWrapper) -> torch.Tensor: + residual = self._forward_embed(wrapped_batch) + + residual, hidden_states = self.norm(residual, + None, + gamma=self._transformer[0].ln_gamma, + beta=self._transformer[0].ln_beta) + + for layer_idx in range(self.num_layers): + residual, hidden_states = self._forward_transformer_layer(layer_idx, residual, hidden_states, + wrapped_batch) + + return self._forward_unembed(residual, wrapped_batch) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/policy.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/policy.py new file mode 100644 index 0000000000000000000000000000000000000000..4b081a8e61bde9304c8f1bee920bb8729f1c6aef --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi/policy.py @@ -0,0 +1,32 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any + +from ...config_v2 import RaggedInferenceEngineConfig +from ..inference_policy_base import ContainerMap, InferenceV2Policy +from .containers import PhiNonTransformerContainer, PhiTransformerContainer +from .model import PhiInferenceModel + + +class PhiPolicy(InferenceV2Policy): + + def instantiate_model(self, engine_config: RaggedInferenceEngineConfig, mp_group: Any) -> PhiInferenceModel: + return PhiInferenceModel(config=self._model_config, engine_config=engine_config, base_mp_group=mp_group) + + def build_container_map(self) -> ContainerMap: + map = ContainerMap() + + trans_container_cls = PhiTransformerContainer + transformer_containers = [trans_container_cls(self.model) for _ in range(self.model.num_layers)] + + map.set_transformer_params(['model.layers'], transformer_containers) + + map.set_non_transformer_params(PhiNonTransformerContainer(self.model)) + + map.set_unmapped_params( + [f'model.layers.{i}.self_attn.rotary_emb.inv_freq' for i in range(self.model.num_layers)]) + + return map diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..1a4b756d210c5e336cec1053e6c0533f69afd461 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .policy import Phi3Policy diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..79187e2d2e500c647425dd063d0750127e60dbd8 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/containers.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/containers.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7d7c7f32a0e59e54699db2db96dd848a37dbd486 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/containers.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/model.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/model.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..984bd1398cf95d2635b63fcb5db2ca4014bd59bb Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/model.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/policy.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/policy.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..484713df07452d1d27c4e4574b1b5916f7832baa Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/__pycache__/policy.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/containers.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/containers.py new file mode 100644 index 0000000000000000000000000000000000000000..1cb52a75ae0b8a6c26a92db651ba881509df9405 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/containers.py @@ -0,0 +1,75 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# Create a container object to save model-specific tensors using the policy file above. + +from ..common_parameters import * +from ..layer_container_base import LayerContainer +''' + # HF Phi-3 model looks like this: + +Phi3ForCausalLM( + (model): Phi3Model( + (embed_tokens): Embedding(32064, 3072) + (embed_dropout): Dropout(p=0.0, inplace=False) + (layers): ModuleList( + (0-31): 32 x Phi3DecoderLayer( + (self_attn): Phi3Attention( + (o_proj): Linear(in_features=3072, out_features=3072, bias=False) + (qkv_proj): Linear(in_features=3072, out_features=9216, bias=False) + (rotary_emb): Phi3RotaryEmbedding() + ) + (mlp): PhiMLP( + (gate_up_proj): Linear(in_features=3072, out_features=16384, bias=False) + (down_proj): Linear(in_features=16384, out_features=3072, bias=False) + (activation_fn): SiLU() + ) + (input_layernorm): Phi3RMSNorm((3072,), eps=1e-05) + (resid_attn_dropout): Dropout(p=0.0) + (resid_mlp_dropout): Dropout(p=0.0) + (post_attention_layernorm): Phi3RMSNorm((3072,), eps=1e-05) + ) + ) + (final_layernorm): Phi3RMSNorm((3072,), eps=1e-05) + ) + (lm_head): Linear(in_features=3072, out_features=32064, bias=False) +) +''' + + +class Phi3TransformerContainer(LayerContainer): + """ + Transformer layer container for the Phi model. + """ + qkv_w: FusedQKVParameter + attn_out_w: AttentionOutputParameter + mlp_1_w: FusedGatedMLPParameter + mlp_2_w: MLP2Parameter + attn_norm_gamma: NormParameter + mlp_norm_gamma: NormParameter + + PARAM_MAPPING = { + "self_attn.qkv_proj.weight": "qkv_w.params", + "self_attn.o_proj.weight": "attn_out_w.params", + "mlp.gate_up_proj.weight": "mlp_1_w.params", + "mlp.down_proj.weight": "mlp_2_w.params", + "input_layernorm.weight": "attn_norm_gamma.params", + "post_attention_layernorm.weight": "mlp_norm_gamma.params", + } + + +class Phi3NonTransformerContainer(LayerContainer): + """ + Non-Transformer layer container for the Phi model. + """ + word_emb: EmbeddingParameter + word_unembed_w: UnembedParameter + final_norm_gamma: NormParameter + + PARAM_MAPPING = { + "model.embed_tokens.weight": "word_emb.params", + "model.norm.weight": "final_norm_gamma.params", + "lm_head.weight": "word_unembed_w.params", + } diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/model.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/model.py new file mode 100644 index 0000000000000000000000000000000000000000..507bb4fc9af1a7755b281c3df5b579c00abb98ff --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/model.py @@ -0,0 +1,204 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Optional, Tuple + +import torch + +import deepspeed.comm as dist + +from ...allocator import empty_from +from ...inference_utils import ActivationType, DtypeEnum +from .. import * +from ...modules.configs import * +from ...modules.interfaces import * +from ...ragged import RaggedBatchWrapper + +from .containers import Phi3NonTransformerContainer, Phi3TransformerContainer + + +class Phi3InferenceModel(DSTransformerModelBase): + """ + Inference model implementation for ragged batching for Llama-2 models. + """ + + _non_transformer: Optional[Phi3NonTransformerContainer] + """ + Embed + unembed container. Specializing the type annotation. + """ + + _transformer: Optional[Iterable[Phi3TransformerContainer]] + """ + Per-layer transformer container. Specializing the type annotation. + """ + """ + Properties inherited from `DSInferenceModelBase` + """ + + @property + def max_sequence_length(self) -> int: + return self._config.max_seq_length + + """ + Properties inherited from `DSTransformerModelBase` + """ + + @property + def num_layers(self) -> int: + return self._config.num_hidden_layers + + @property + def model_dim(self) -> int: + return self._config.hidden_size + + @property + def vocab_size(self) -> int: + return self._config.vocab_size + + @property + def head_size(self) -> int: + return self.model_dim // self.n_heads + + @property + def n_heads(self) -> int: + return self._config.num_attention_heads + + @property + def intermediate_dim(self) -> int: + return self._config.intermediate_size + + @property + def n_heads_kv(self) -> int: + return self._config.num_key_value_heads + + @property + def activation_dtype(self) -> DtypeEnum: + if self._config.torch_dtype == torch.float16: + return DtypeEnum.fp16 + elif self._config.torch_dtype == torch.bfloat16: + return DtypeEnum.bf16 + else: + raise NotImplementedError("Only fp16 and bf16 are supported") + + @property + def mlp_activation_fn(self) -> ActivationType: + activation = self._config.hidden_act.lower() + if activation == "gelu": + return ActivationType.GEGLU + elif activation == "relu": + return ActivationType.ReGLU + elif activation == "gegelu": + return ActivationType.GEGLU + elif activation == "silu": + return ActivationType.SiGLU + else: + raise NotImplementedError(f"Activation {activation} not supported") + + @property + def norm_type(self) -> NormTypeEnum: + return NormTypeEnum.RMSNorm + + @property + def positional_embedding_type(self) -> PositionalEmbeddingType: + return PositionalEmbeddingType.rotate_half + + @property + def positional_embedding_config(self) -> Optional[RotateHalfConfig]: + return RotateHalfConfig(theta_base=self._config.rope_theta) + + """ + Forward implementations + """ + + def _forward_embed(self, ragged_batch: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs the embedding lookup prior to running the transformer of the model. + + Arguments: + ragged_batch (RaggedBatchWrapper): The batch to embed. + + Returns: + torch.Tensor: The embedded batch. + """ + embed = self.embed(ragged_batch, self._non_transformer.word_emb) + + if embed.shape[-1] != self.model_dim: + raise ValueError(f"Embedding output shape {embed.shape} does not match model_dim {self.model_dim}") + + return embed + + def _forward_transformer_layer(self, layer_idx: int, residual: torch.Tensor, hidden_states: torch.Tensor, + ragged_batch_info: RaggedBatchWrapper) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Executes one (slightly offset) layer of the transformer. This implementation does a peak-ahead + optimization to fuse the layer norm of the next layer into the current layer. + + Arguments: + layer_idx (int): The index of the layer to execute. + residual (torch.Tensor): The residual tensor from the previous layer. + hidden_states (torch.Tensor): The hidden states from the previous layer. This is the + hidden states after pre normalization. + ragged_batch_info (RaggedBatchWrapper): The batch metadata. + """ + cur_params = self._transformer[layer_idx] + kv_cache = self.state_manager.get_cache(layer_idx) + + hidden_states = self.qkv(hidden_states, cur_params.qkv_w, b=None) + hidden_states = self.attn(hidden_states, kv_cache, ragged_batch_info) + hidden_states = self.attn_out(hidden_states, cur_params.attn_out_w, b=None) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + residual, hidden_states = self.norm(residual, hidden_states, cur_params.mlp_norm_gamma, beta=None) + + hidden_states = self.mlp_1(hidden_states, cur_params.mlp_1_w, b=None) + hidden_states = self.mlp_2(hidden_states, cur_params.mlp_2_w, b=None) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + if layer_idx != self.num_layers - 1: + next_params = self._transformer[layer_idx + 1] + residual, hidden_states = self.norm(residual, hidden_states, next_params.attn_norm_gamma, beta=None) + else: + # On last layer, we just need to perform the residual add. Adding into the residual + # here is safe. + residual.add_(hidden_states) + + return residual, hidden_states + + def _forward_unembed(self, hidden_states: torch.Tensor, ragged_batch_info: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs unembedding of the hidden states to logits. This will only sample the final + token of each sequence. + """ + logits = self.unembed(hidden_states, + self._non_transformer.word_unembed_w, + ragged_batch_info, + gamma=self._non_transformer.final_norm_gamma) + + if self.tp_size > 1: + comm_buffer = empty_from(self._comm_logits, (self.tp_size, logits.shape[0], logits.shape[1])) + full_logits = empty_from(self._return_logits, (logits.shape[0], self.vocab_size)) + + dist.all_gather_into_tensor(comm_buffer, logits, group=self._base_mp_group) + + full_logits.copy_(comm_buffer.permute(1, 0, 2).reshape(logits.shape[0], self.vocab_size)) + + return full_logits + else: + return logits + + def forward(self, wrapped_batch: RaggedBatchWrapper) -> torch.Tensor: + residual = self._forward_embed(wrapped_batch) + + residual, hidden_states = self.norm(residual, None, gamma=self._transformer[0].attn_norm_gamma, beta=None) + + for layer_idx in range(self.num_layers): + residual, hidden_states = self._forward_transformer_layer(layer_idx, residual, hidden_states, + wrapped_batch) + + return self._forward_unembed(residual, wrapped_batch) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/policy.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/policy.py new file mode 100644 index 0000000000000000000000000000000000000000..a1b445929053afb53e6998cd11f176b51651baa5 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/phi3/policy.py @@ -0,0 +1,30 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any + +from ...config_v2 import RaggedInferenceEngineConfig +from ..inference_policy_base import ContainerMap, InferenceV2Policy +from .containers import Phi3NonTransformerContainer, Phi3TransformerContainer +from .model import Phi3InferenceModel + + +class Phi3Policy(InferenceV2Policy): + + def instantiate_model(self, engine_config: RaggedInferenceEngineConfig, mp_group: Any) -> Phi3InferenceModel: + return Phi3InferenceModel(config=self._model_config, engine_config=engine_config, base_mp_group=mp_group) + + def build_container_map(self) -> ContainerMap: + map = ContainerMap() + + transformer_containers = [Phi3TransformerContainer(self.model) for _ in range(self.model.num_layers)] + + map.set_transformer_params(['model.layers'], transformer_containers) + + map.set_non_transformer_params(Phi3NonTransformerContainer(self.model)) + + map.set_unmapped_params([]) + + return map diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..18206048fa299b8ced074aba9ea7f5212d3e0e3b --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .policy import QwenPolicy diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..72228cf402b586433259bd8dbc23b243937a8ea0 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/container.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/container.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2147eb77c869ccc7de41237b3b0e040a30877979 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/container.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/model.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/model.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..cc7ea434dae0b55db9595a5198e0d872ac4e1920 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/model.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/policy.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/policy.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b0caea1623c8fa88f496327b409587d222f26c1c Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/__pycache__/policy.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/container.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/container.py new file mode 100644 index 0000000000000000000000000000000000000000..313de68555b90f1dd38935de286d93f9cbd8b382 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/container.py @@ -0,0 +1,77 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# Create a container object to save model-specific tensors using the policy file above. + +from ..common_parameters import * +from ..layer_container_base import LayerContainer +''' + # HF Qwen model looks like this: + +QWenLMHeadModel( + (transformer): QWenModel( + (wte): Embedding(151936, 4096) + (drop): Dropout(p=0.0, inplace=False) + (rotary_emb): RotaryEmbedding() + (h): ModuleList( + (0-31): 32 x QWenBlock( + (ln_1): RMSNorm() + (attn): QWenAttention( + (c_attn): Linear(in_features=4096, out_features=12288, bias=True) + (c_proj): Linear(in_features=4096, out_features=4096, bias=False) + (attn_dropout): Dropout(p=0.0, inplace=False) + ) + (ln_2): RMSNorm() + (mlp): QWenMLP( + (w1): Linear(in_features=4096, out_features=11008, bias=False) + (w2): Linear(in_features=4096, out_features=11008, bias=False) + (c_proj): Linear(in_features=11008, out_features=4096, bias=False) + ) + ) + ) + (ln_f): RMSNorm() + ) + (lm_head): Linear(in_features=4096, out_features=151936, bias=False) +) +''' + + +class QwenTransformerContainer(LayerContainer): + """ + Transformer layer container for the Qwen model. + """ + qkv_w: FusedQKVParameter + qkv_b: FusedQKVParameter + attn_out_w: AttentionOutputParameter + mlp_1_w: GatedMLPParameter + mlp_2_w: MLP2Parameter + attn_norm_gamma: NormParameter + mlp_norm_gamma: NormParameter + + PARAM_MAPPING = { + "attn.c_attn.weight": "qkv_w.params", + "attn.c_attn.bias": "qkv_b.params", + "attn.c_proj.weight": "attn_out_w.params", + "mlp.w1.weight": "mlp_1_w.up_params", + "mlp.w2.weight": "mlp_1_w.gate_params", + "mlp.c_proj.weight": "mlp_2_w.params", + "ln_1.weight": "attn_norm_gamma.params", + "ln_2.weight": "mlp_norm_gamma.params", + } + + +class QwenNonTransformerContainer(LayerContainer): + """ + Non-Transformer layer container for the Qwen model. + """ + word_emb: EmbeddingParameter + word_unembed: UnembedParameter + final_norm: NormParameter + + PARAM_MAPPING = { + "transformer.wte.weight": "word_emb.params", + "transformer.ln_f.weight": "final_norm.params", + "lm_head.weight": "word_unembed.params", + } diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/model.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/model.py new file mode 100644 index 0000000000000000000000000000000000000000..e867e4be67133cb512c737db3a3c45cb2294e404 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/model.py @@ -0,0 +1,223 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Optional, Tuple + +import torch + +import deepspeed.comm as dist + +from ...allocator import empty_from +from ...inference_utils import ActivationType, DtypeEnum +from .. import * +from ...modules.configs import * +from ...modules.interfaces import * +from ...modules import heuristics +from ...ragged import RaggedBatchWrapper + +from .container import QwenNonTransformerContainer, QwenTransformerContainer + + +class QwenInferenceModel(DSTransformerModelBase): + """ + Inference model implementation for ragged batching for Llama-2 models. + """ + + _non_transformer: Optional[QwenNonTransformerContainer] + """ + Embed + unembed container. Specializing the type annotation. + """ + + _transformer: Optional[Iterable[QwenTransformerContainer]] + """ + Per-layer transformer container. Specializing the type annotation. + """ + """ + Properties ineherited from `DSInferenceModelBase` + """ + + @property + def max_sequence_length(self) -> int: + return self._config.max_seq_length + + """ + Properties ineherited from `DSTransformerModelBase` + """ + + @property + def num_layers(self) -> int: + return self._config.num_hidden_layers + + @property + def model_dim(self) -> int: + return self._config.hidden_size + + @property + def vocab_size(self) -> int: + return self._config.vocab_size + + @property + def head_size(self) -> int: + return self.model_dim // self.n_heads + + @property + def n_heads(self) -> int: + return self._config.num_attention_heads + + @property + def intermediate_dim(self) -> int: + return self._config.intermediate_size // 2 + + @property + def n_heads_kv(self) -> int: + return self._config.hidden_size // self._config.kv_channels + + @property + def activation_dtype(self) -> DtypeEnum: + autoset_precision = self._config.bf16 + self._config.fp16 == 0 + if autoset_precision: + return DtypeEnum.fp16 + if self._config.fp16: + return DtypeEnum.fp16 + elif self._config.bf16: + # TODO(ZonePG): bf16 inference results may be different from huggingface bf16, + # because in rms_norm, Qwen still use float() instead of bf16 + return DtypeEnum.bf16 + else: + raise NotImplementedError("Only fp16 and bf16 are supported") + + @property + def mlp_activation_fn(self) -> ActivationType: + return ActivationType.SiGLU + + @property + def norm_type(self) -> NormTypeEnum: + return NormTypeEnum.RMSNorm + + @property + def positional_embedding_type(self) -> PositionalEmbeddingType: + return PositionalEmbeddingType.rotate_half + + @property + def positional_embedding_config(self) -> Optional[RotateHalfConfig]: + return RotateHalfConfig(theta_base=self._config.rotary_emb_base) + + def make_norm_layer(self) -> None: + """ + Instantiates the normalization layer for the model. This sets the `self.norm` attribute. + + TODO(cmikeh2): In the future we'll distinguish between the different norm objects, + but for now we'll just use the same one for all of them. + """ + norm_config = DSNormConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + type=self.norm_type, + channels=self.model_dim, + residual_dtype=self.activation_dtype, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + eps=self._config.layer_norm_epsilon, + ) + + self.norm = heuristics.instantiate_pre_norm(norm_config, self._engine_config) + + """ + Forward implementations + """ + + def _forward_embed(self, ragged_batch: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs the embedding lookup prior to running the transformer of the model. + + Arguments: + ragged_batch (RaggedBatchWrapper): The batch to embed. + + Returns: + torch.Tensor: The embedded batch. + """ + embed = self.embed(ragged_batch, self._non_transformer.word_emb) + + if embed.shape[-1] != self.model_dim: + raise ValueError(f"Embedding output shape {embed.shape} does not match model_dim {self.model_dim}") + + return embed + + def _forward_transformer_layer(self, layer_idx: int, residual: torch.Tensor, hidden_states: torch.Tensor, + ragged_batch_info: RaggedBatchWrapper) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Executes one (slightly offset) layer of the transformer. This implementation does a peak-ahead + optimization to fuse the layer norm of the next layer into the current layer. + + Arguments: + layer_idx (int): The index of the layer to execute. + residual (torch.Tensor): The residual tensor from the previous layer. + hidden_states (torch.Tensor): The hidden states from the previous layer. This is the + hidden states after pre normalization. + ragged_batch_info (RaggedBatchWrapper): The batch metadata. + """ + # TODO(cmikeh2): Distribute ragged_batch_info to all modules + + cur_params = self._transformer[layer_idx] + kv_cache = self.state_manager.get_cache(layer_idx) + + hidden_states = self.qkv(hidden_states, cur_params.qkv_w, b=cur_params.qkv_b) + hidden_states = self.attn(hidden_states, kv_cache, ragged_batch_info) + hidden_states = self.attn_out(hidden_states, cur_params.attn_out_w, b=None) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + residual, hidden_states = self.norm(residual, hidden_states, cur_params.mlp_norm_gamma, beta=None) + + # Should be configurable in the future + hidden_states = self.mlp_1(hidden_states, cur_params.mlp_1_w, b=None) + hidden_states = self.mlp_2(hidden_states, cur_params.mlp_2_w, b=None) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + if layer_idx != self.num_layers - 1: + next_params = self._transformer[layer_idx + 1] + residual, hidden_states = self.norm(residual, hidden_states, next_params.attn_norm_gamma, beta=None) + else: + # On last layer, we just need to perform the residual add. Adding into the residual + # here is safe. + residual.add_(hidden_states) + + return residual, hidden_states + + def _forward_unembed(self, hidden_states: torch.Tensor, ragged_batch_info: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs unembedding of the hidden states to logits. This will only sample the final + token of each sequence. + """ + logits = self.unembed(hidden_states, + self._non_transformer.word_unembed, + ragged_batch_info, + gamma=self._non_transformer.final_norm) + + if self.tp_size > 1: + comm_buffer = empty_from(self._comm_logits, (self.tp_size, logits.shape[0], logits.shape[1])) + full_logits = empty_from(self._return_logits, (logits.shape[0], self.vocab_size)) + + dist.all_gather_into_tensor(comm_buffer, logits, group=self._base_mp_group) + + full_logits.copy_(comm_buffer.permute(1, 0, 2).reshape(logits.shape[0], self.vocab_size)) + + return full_logits + else: + return logits + + def forward(self, wrapped_batch: RaggedBatchWrapper) -> torch.Tensor: + + residual = self._forward_embed(wrapped_batch) + + residual, hidden_states = self.norm(residual, None, self._transformer[0].attn_norm_gamma, beta=None) + + for layer_idx in range(self.num_layers): + residual, hidden_states = self._forward_transformer_layer(layer_idx, residual, hidden_states, + wrapped_batch) + + return self._forward_unembed(residual, wrapped_batch) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/policy.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/policy.py new file mode 100644 index 0000000000000000000000000000000000000000..a9263f553621ec715cc5527e95034b842be0132b --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen/policy.py @@ -0,0 +1,30 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any + +from ...config_v2 import RaggedInferenceEngineConfig +from ..inference_policy_base import ContainerMap, InferenceV2Policy +from .container import QwenNonTransformerContainer, QwenTransformerContainer +from .model import QwenInferenceModel + + +class QwenPolicy(InferenceV2Policy): + + def instantiate_model(self, engine_config: RaggedInferenceEngineConfig, mp_group: Any) -> QwenInferenceModel: + return QwenInferenceModel(config=self._model_config, engine_config=engine_config, base_mp_group=mp_group) + + def build_container_map(self) -> ContainerMap: + map = ContainerMap() + + transformer_containers = [QwenTransformerContainer(self.model) for _ in range(self.model.num_layers)] + + map.set_transformer_params(['transformer.h'], transformer_containers) + + map.set_non_transformer_params(QwenNonTransformerContainer(self.model)) + + map.set_unmapped_params(['transformer.rotary_emb.inv_freq']) + + return map diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..80b09757c74db181e5a7729579c89d530f65f25c --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .policy import Qwen2Policy diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..4a09a05e9ae2cf934416855c1de85db6d7834497 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/container.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/container.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..fd92da10b9bd610ee01c98b6c5808a07563412ab Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/container.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/model.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/model.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c6c05b7a6ed55bb808d9ca2b5196ce78b913d148 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/model.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/policy.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/policy.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..014bbdb256f48eef278fa5d967193d6c6428616a Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/__pycache__/policy.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/container.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/container.py new file mode 100644 index 0000000000000000000000000000000000000000..6556d87d6afb2c2d36781d9af8c4a09ce035aee1 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/container.py @@ -0,0 +1,82 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# Create a container object to save model-specific tensors using the policy file above. + +from ..common_parameters import * +from ..layer_container_base import LayerContainer +''' + # HF Qwen2 model looks like this: + +Qwen2ForCausalLM( + (model): Qwen2Model( + (embed_tokens): Embedding(151936, 1024) + (layers): ModuleList( + (0-23): 24 x Qwen2DecoderLayer( + (self_attn): Qwen2SdpaAttention( + (q_proj): Linear(in_features=1024, out_features=1024, bias=True) + (k_proj): Linear(in_features=1024, out_features=1024, bias=True) + (v_proj): Linear(in_features=1024, out_features=1024, bias=True) + (o_proj): Linear(in_features=1024, out_features=1024, bias=False) + (rotary_emb): Qwen2RotaryEmbedding() + ) + (mlp): Qwen2MLP( + (gate_proj): Linear(in_features=1024, out_features=2816, bias=False) + (up_proj): Linear(in_features=1024, out_features=2816, bias=False) + (down_proj): Linear(in_features=2816, out_features=1024, bias=False) + (act_fn): SiLU() + ) + (input_layernorm): Qwen2RMSNorm() + (post_attention_layernorm): Qwen2RMSNorm() + ) + ) + (norm): Qwen2RMSNorm() + ) + (lm_head): Linear(in_features=1024, out_features=151936, bias=False) +) +''' + + +class Qwen2TransformerContainer(LayerContainer): + """ + Transformer layer container for the Qwen2 model. + """ + qkv_w: UnfusedQKVParameter + qkv_b: UnfusedQKVParameter + attn_out_w: AttentionOutputParameter + mlp_1_w: GatedMLPParameter + mlp_2_w: MLP2Parameter + attn_norm_gamma: NormParameter + mlp_norm_gamma: NormParameter + + PARAM_MAPPING = { + "self_attn.q_proj.weight": "qkv_w.q_params", + "self_attn.k_proj.weight": "qkv_w.k_params", + "self_attn.v_proj.weight": "qkv_w.v_params", + "self_attn.q_proj.bias": "qkv_b.q_params", + "self_attn.k_proj.bias": "qkv_b.k_params", + "self_attn.v_proj.bias": "qkv_b.v_params", + "self_attn.o_proj.weight": "attn_out_w.params", + "mlp.gate_proj.weight": "mlp_1_w.gate_params", + "mlp.up_proj.weight": "mlp_1_w.up_params", + "mlp.down_proj.weight": "mlp_2_w.params", + "input_layernorm.weight": "attn_norm_gamma.params", + "post_attention_layernorm.weight": "mlp_norm_gamma.params", + } + + +class Qwen2NonTransformerContainer(LayerContainer): + """ + Non-Transformer layer container for the Qwen2 model. + """ + word_emb: EmbeddingParameter + word_unembed: UnembedParameter + final_norm: NormParameter + + PARAM_MAPPING = { + "model.embed_tokens.weight": "word_emb.params", + "model.norm.weight": "final_norm.params", + "lm_head.weight": "word_unembed.params", + } diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/model.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/model.py new file mode 100644 index 0000000000000000000000000000000000000000..d535462a954d4156ff32f2d3667a6a17713b4871 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/model.py @@ -0,0 +1,221 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Optional, Tuple + +import torch + +import deepspeed.comm as dist + +from ...allocator import empty_from +from ...inference_utils import ActivationType, DtypeEnum +from .. import * +from ...modules.configs import * +from ...modules.interfaces import * +from ...modules import heuristics +from ...ragged import RaggedBatchWrapper + +from .container import Qwen2NonTransformerContainer, Qwen2TransformerContainer + + +class Qwen2InferenceModel(DSTransformerModelBase): + """ + Inference model implementation for ragged batching for Llama-2 models. + """ + + _non_transformer: Optional[Qwen2NonTransformerContainer] + """ + Embed + unembed container. Specializing the type annotation. + """ + + _transformer: Optional[Iterable[Qwen2TransformerContainer]] + """ + Per-layer transformer container. Specializing the type annotation. + """ + """ + Properties ineherited from `DSInferenceModelBase` + """ + + @property + def max_sequence_length(self) -> int: + return self._config.max_seq_length + + """ + Properties ineherited from `DSTransformerModelBase` + """ + + @property + def num_layers(self) -> int: + return self._config.num_hidden_layers + + @property + def model_dim(self) -> int: + return self._config.hidden_size + + @property + def vocab_size(self) -> int: + return self._config.vocab_size + + @property + def head_size(self) -> int: + return self.model_dim // self.n_heads + + @property + def n_heads(self) -> int: + return self._config.num_attention_heads + + @property + def intermediate_dim(self) -> int: + return self._config.intermediate_size + + @property + def n_heads_kv(self) -> int: + return self._config.num_key_value_heads + + @property + def activation_dtype(self) -> DtypeEnum: + # TODO(ZonePG): bf16 inference results may be different from huggingface bf16, + # because in rms_norm, Qwen still use float() instead of bf16 + # if self._config.torch_dtype == torch.float16: + # return DtypeEnum.fp16 + # elif self._config.torch_dtype == torch.bfloat16: + # return DtypeEnum.bf16 + # else: + # raise NotImplementedError("Only fp16 and bf16 are supported") + return DtypeEnum.fp16 + + @property + def mlp_activation_fn(self) -> ActivationType: + return ActivationType.SiGLU + + @property + def norm_type(self) -> NormTypeEnum: + return NormTypeEnum.RMSNorm + + @property + def positional_embedding_type(self) -> PositionalEmbeddingType: + return PositionalEmbeddingType.rotate_half + + @property + def positional_embedding_config(self) -> Optional[RotateHalfConfig]: + return RotateHalfConfig(theta_base=self._config.rope_theta) + + def make_norm_layer(self) -> None: + """ + Instantiates the normalization layer for the model. This sets the `self.norm` attribute. + + TODO(cmikeh2): In the future we'll distinguish between the different norm objects, + but for now we'll just use the same one for all of them. + """ + norm_config = DSNormConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + type=self.norm_type, + channels=self.model_dim, + residual_dtype=self.activation_dtype, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + eps=self._config.rms_norm_eps, + ) + + self.norm = heuristics.instantiate_pre_norm(norm_config, self._engine_config) + + """ + Forward implementations + """ + + def _forward_embed(self, ragged_batch: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs the embedding lookup prior to running the transformer of the model. + + Arguments: + ragged_batch (RaggedBatchWrapper): The batch to embed. + + Returns: + torch.Tensor: The embedded batch. + """ + embed = self.embed(ragged_batch, self._non_transformer.word_emb) + + if embed.shape[-1] != self.model_dim: + raise ValueError(f"Embedding output shape {embed.shape} does not match model_dim {self.model_dim}") + + return embed + + def _forward_transformer_layer(self, layer_idx: int, residual: torch.Tensor, hidden_states: torch.Tensor, + ragged_batch_info: RaggedBatchWrapper) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Executes one (slightly offset) layer of the transformer. This implementation does a peak-ahead + optimization to fuse the layer norm of the next layer into the current layer. + + Arguments: + layer_idx (int): The index of the layer to execute. + residual (torch.Tensor): The residual tensor from the previous layer. + hidden_states (torch.Tensor): The hidden states from the previous layer. This is the + hidden states after pre normalization. + ragged_batch_info (RaggedBatchWrapper): The batch metadata. + """ + # TODO(cmikeh2): Distribute ragged_batch_info to all modules + + cur_params = self._transformer[layer_idx] + kv_cache = self.state_manager.get_cache(layer_idx) + + hidden_states = self.qkv(hidden_states, cur_params.qkv_w, b=cur_params.qkv_b) + hidden_states = self.attn(hidden_states, kv_cache, ragged_batch_info) + hidden_states = self.attn_out(hidden_states, cur_params.attn_out_w, b=None) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + residual, hidden_states = self.norm(residual, hidden_states, cur_params.mlp_norm_gamma, beta=None) + + # Should be configurable in the future + hidden_states = self.mlp_1(hidden_states, cur_params.mlp_1_w, b=None) + hidden_states = self.mlp_2(hidden_states, cur_params.mlp_2_w, b=None) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + if layer_idx != self.num_layers - 1: + next_params = self._transformer[layer_idx + 1] + residual, hidden_states = self.norm(residual, hidden_states, next_params.attn_norm_gamma, beta=None) + else: + # On last layer, we just need to perform the residual add. Adding into the residual + # here is safe. + residual.add_(hidden_states) + + return residual, hidden_states + + def _forward_unembed(self, hidden_states: torch.Tensor, ragged_batch_info: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs unembedding of the hidden states to logits. This will only sample the final + token of each sequence. + """ + logits = self.unembed(hidden_states, + self._non_transformer.word_unembed, + ragged_batch_info, + gamma=self._non_transformer.final_norm) + + if self.tp_size > 1: + comm_buffer = empty_from(self._comm_logits, (self.tp_size, logits.shape[0], logits.shape[1])) + full_logits = empty_from(self._return_logits, (logits.shape[0], self.vocab_size)) + + dist.all_gather_into_tensor(comm_buffer, logits, group=self._base_mp_group) + + full_logits.copy_(comm_buffer.permute(1, 0, 2).reshape(logits.shape[0], self.vocab_size)) + + return full_logits + else: + return logits + + def forward(self, wrapped_batch: RaggedBatchWrapper) -> torch.Tensor: + + residual = self._forward_embed(wrapped_batch) + + residual, hidden_states = self.norm(residual, None, self._transformer[0].attn_norm_gamma, beta=None) + + for layer_idx in range(self.num_layers): + residual, hidden_states = self._forward_transformer_layer(layer_idx, residual, hidden_states, + wrapped_batch) + + return self._forward_unembed(residual, wrapped_batch) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/policy.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/policy.py new file mode 100644 index 0000000000000000000000000000000000000000..9c5db2ba0065e0b545464c3ffe89095cb0e03148 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2/policy.py @@ -0,0 +1,31 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any + +from ...config_v2 import RaggedInferenceEngineConfig +from ..inference_policy_base import ContainerMap, InferenceV2Policy +from .container import Qwen2NonTransformerContainer, Qwen2TransformerContainer +from .model import Qwen2InferenceModel + + +class Qwen2Policy(InferenceV2Policy): + + def instantiate_model(self, engine_config: RaggedInferenceEngineConfig, mp_group: Any) -> Qwen2InferenceModel: + return Qwen2InferenceModel(config=self._model_config, engine_config=engine_config, base_mp_group=mp_group) + + def build_container_map(self) -> ContainerMap: + map = ContainerMap() + + transformer_containers = [Qwen2TransformerContainer(self.model) for _ in range(self.model.num_layers)] + + map.set_transformer_params(['model.layers'], transformer_containers) + + map.set_non_transformer_params(Qwen2NonTransformerContainer(self.model)) + + map.set_unmapped_params( + [f'model.layers.{i}.self_attn.rotary_emb.inv_freq' for i in range(self.model.num_layers)]) + + return map diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..23e06a7700237f1e8e7c91f83c6f2986f2db8993 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .policy import Qwen2MoePolicy diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..23d8654f6b9a21adc35e9e181c3dd19d777be3de Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/container.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/container.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..9f1ba598a5dbfa011393f23fcd7929de38876df8 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/container.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/model.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/model.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..66ea6fefc76bc96cba0952430a9ff362c5f2a69e Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/model.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/policy.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/policy.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..fec3e9bb16ced56966bd36f814e77d7460cc8566 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/__pycache__/policy.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/container.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/container.py new file mode 100644 index 0000000000000000000000000000000000000000..e499379da7e368ccb9279333bd4e28ec8cff7e0d --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/container.py @@ -0,0 +1,103 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# Create a container object to save model-specific tensors using the policy file above. + +from ..common_parameters import * +from ..layer_container_base import LayerContainer +''' + # HF Qwen2-57B-A14B model looks like this: + +Qwen2MoeForCausalLM( + (model): Qwen2MoeModel( + (embed_tokens): Embedding(151936, 3584) + (layers): ModuleList( + (0-27): 28 x Qwen2MoeDecoderLayer( + (self_attn): Qwen2MoeSdpaAttention( + (q_proj): Linear(in_features=3584, out_features=3584, bias=True) + (k_proj): Linear(in_features=3584, out_features=512, bias=True) + (v_proj): Linear(in_features=3584, out_features=512, bias=True) + (o_proj): Linear(in_features=3584, out_features=3584, bias=False) + (rotary_emb): Qwen2MoeRotaryEmbedding() + ) + (mlp): Qwen2MoeSparseMoeBlock( + (gate): Linear(in_features=3584, out_features=64, bias=False) + (experts): ModuleList( + (0-63): 64 x Qwen2MoeMLP( + (gate_proj): Linear(in_features=3584, out_features=2560, bias=False) + (up_proj): Linear(in_features=3584, out_features=2560, bias=False) + (down_proj): Linear(in_features=2560, out_features=3584, bias=False) + (act_fn): SiLU() + ) + ) + (shared_expert): Qwen2MoeMLP( + (gate_proj): Linear(in_features=3584, out_features=20480, bias=False) + (up_proj): Linear(in_features=3584, out_features=20480, bias=False) + (down_proj): Linear(in_features=20480, out_features=3584, bias=False) + (act_fn): SiLU() + ) + (shared_expert_gate): Linear(in_features=3584, out_features=1, bias=False) + ) + (input_layernorm): Qwen2MoeRMSNorm((3584,), eps=1e-06) + (post_attention_layernorm): Qwen2MoeRMSNorm((3584,), eps=1e-06) + ) + ) + (norm): Qwen2MoeRMSNorm((3584,), eps=1e-06) + ) + (lm_head): Linear(in_features=3584, out_features=151936, bias=False) +) +''' + + +class Qwen2MoeTransformerContainer(LayerContainer): + """ + Transformer layer container for the Qwen2Moe model. + """ + qkv_w: UnfusedQKVParameter + qkv_b: UnfusedQKVParameter + attn_out_w: AttentionOutputParameter + moe_gate: MoEGatingWeightParameter + moe_mlp_1: UnfusedMoEGatedMLPParameter + moe_mlp_2: UnfusedMoEMLP2Parameter + shared_moe_mlp_1: GatedMLPParameter + shared_moe_mlp_2: MLP2Parameter + shared_moe_gate: MoEGatingWeightParameter + attn_norm_gamma: NormParameter + mlp_norm_gamma: NormParameter + + PARAM_MAPPING = { + "self_attn.q_proj.weight": "qkv_w.q_params", + "self_attn.k_proj.weight": "qkv_w.k_params", + "self_attn.v_proj.weight": "qkv_w.v_params", + "self_attn.q_proj.bias": "qkv_b.q_params", + "self_attn.k_proj.bias": "qkv_b.k_params", + "self_attn.v_proj.bias": "qkv_b.v_params", + "self_attn.o_proj.weight": "attn_out_w.params", + "mlp.gate.weight": "moe_gate.params", + "mlp.experts.*.gate_proj.weight": "moe_mlp_1.gating_experts", + "mlp.experts.*.up_proj.weight": "moe_mlp_1.up_experts", + "mlp.experts.*.down_proj.weight": "moe_mlp_2.experts", + "mlp.shared_expert.gate_proj.weight": "shared_moe_mlp_1.gate_params", + "mlp.shared_expert.up_proj.weight": "shared_moe_mlp_1.up_params", + "mlp.shared_expert.down_proj.weight": "shared_moe_mlp_2.params", + "mlp.shared_expert_gate.weight": "shared_moe_gate.params", + "input_layernorm.weight": "attn_norm_gamma.params", + "post_attention_layernorm.weight": "mlp_norm_gamma.params", + } + + +class Qwen2MoeNonTransformerContainer(LayerContainer): + """ + Non-Transformer layer container for the Qwen2Moe model. + """ + word_emb: EmbeddingParameter + word_unembed: UnembedParameter + final_norm: NormParameter + + PARAM_MAPPING = { + "model.embed_tokens.weight": "word_emb.params", + "model.norm.weight": "final_norm.params", + "lm_head.weight": "word_unembed.params", + } diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/model.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/model.py new file mode 100644 index 0000000000000000000000000000000000000000..c7841b24e5fc79c6428422ca07618e78683cd7f1 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/model.py @@ -0,0 +1,359 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Optional, Tuple + +import torch + +import deepspeed.comm as dist + +from ...allocator import empty_from +from ...config_v2 import RaggedInferenceEngineConfig +from ...inference_utils import ActivationType, DtypeEnum +from ...model_implementations import * +from ...modules.configs import * +from ...modules.interfaces import * +from ...modules import heuristics +from ...ragged import RaggedBatchWrapper +from ..inference_model_base import ( + DSModelImplementationConfig, + MPType, +) + +from .container import Qwen2MoeNonTransformerContainer, Qwen2MoeTransformerContainer + + +class Qwen2MoeInferenceModel(DSMoETransformerModelBase): + """ + Inference model implementation for Qwen2MoE models. + """ + + _non_transformer: Optional[Qwen2MoeNonTransformerContainer] + """ + Embed + unembed container. Specializing the type annotation. + """ + + _transformer: Optional[Iterable[Qwen2MoeTransformerContainer]] + """ + Per-layer transformer container. Specializing the type annotation. + """ + """ + Properties ineherited from `DSInferenceModelBase` + """ + + @property + def max_sequence_length(self) -> int: + return self._config.max_position_embeddings + + """ + Properties ineherited from `DSTransformerModelBase` + """ + + @property + def num_layers(self) -> int: + return self._config.num_hidden_layers + + @property + def model_dim(self) -> int: + return self._config.hidden_size + + @property + def vocab_size(self) -> int: + return self._config.vocab_size + + @property + def head_size(self) -> int: + return self.model_dim // self.n_heads + + @property + def n_heads(self) -> int: + return self._config.num_attention_heads + + @property + def intermediate_dim(self) -> int: + return self._config.shared_expert_intermediate_size + + @property + def n_heads_kv(self) -> int: + return self._config.num_key_value_heads + + @property + def activation_dtype(self) -> DtypeEnum: + # TODO(ZonePG): bf16 inference results may be different from huggingface bf16, + # because in rms_norm, Qwen still use float() instead of bf16 + # if self._config.torch_dtype == torch.float16: + # return DtypeEnum.fp16 + # elif self._config.torch_dtype == torch.bfloat16: + # return DtypeEnum.bf16 + # else: + # raise NotImplementedError("Only fp16 and bf16 are supported") + return DtypeEnum.fp16 + + @property + def mlp_activation_fn(self) -> ActivationType: + return ActivationType.SiGLU + + @property + def norm_type(self) -> NormTypeEnum: + return NormTypeEnum.RMSNorm + + @property + def positional_embedding_type(self) -> PositionalEmbeddingType: + return PositionalEmbeddingType.rotate_half + + @property + def positional_embedding_config(self) -> Optional[RotateHalfConfig]: + return RotateHalfConfig(theta_base=self._config.rope_theta) + + """ + Inherited from `DSMoETransformerModelBase` + """ + + @property + def n_experts(self) -> int: + return self._config.num_experts + + @property + def n_top_k(self) -> int: + return self._config.num_experts_per_tok + + @property + def normalize_expert_scores(self) -> bool: + return self._config.norm_topk_prob + + def make_moe_layer(self) -> None: + """ + Instantiates the MoE layer for the model. This sets the `self.moe` attribute. + """ + sharded_dim = sharded_intermediate_dim(self.intermediate_dim // self.n_top_k, self.tp_size, self.tp_rank) + + moe_config = DSMoEConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + model_dim=self.model_dim, + intermediate_features=sharded_dim, + activation=self.mlp_activation_fn, + n_experts=self.n_experts, + top_k=self.n_top_k, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + normalize_scores=self.normalize_expert_scores, + ) + + self.moe = heuristics.instantiate_moe(moe_config, self._engine_config) + + ######### MLP 1 ######### + def make_shared_expert_mlp_1_layer(self) -> None: + """ + Instantiates the linear projection layer for the first MLP in the feedforward network. + This sets the `self.mlp_1` attribute. + """ + shard_size = sharded_intermediate_dim(self.intermediate_dim, self.tp_size, self.tp_rank) + + linear_config = DSLinearConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + in_channels=self.model_dim, + out_channels=shard_size, + activation=self.mlp_activation_fn, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + ) + + self.shared_expert_mlp_1 = heuristics.instantiate_linear(linear_config, self._engine_config) + + ######### MLP 2 ######### + def make_shared_expert_mlp_2_layer(self) -> None: + """ + Instantiates the linear projection layer for the second MLP in the feedforward network. + This sets the `self.mlp_2` attribute. + """ + shard_size = sharded_intermediate_dim(self.intermediate_dim, self.tp_size, self.tp_rank) + + linear_config = DSLinearConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + in_channels=shard_size, + out_channels=self.model_dim, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + ) + + self.shared_expert_mlp_2 = heuristics.instantiate_linear(linear_config, self._engine_config) + + ######### MLP 2 ######### + def make_shared_expert_gate_layer(self) -> None: + """ + Instantiates the linear projection layer for the second MLP in the feedforward network. + This sets the `self.mlp_2` attribute. + """ + shard_size = sharded_intermediate_dim(self.model_dim, self.tp_size, self.tp_rank) + + linear_config = DSLinearConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + in_channels=shard_size, + out_channels=8, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + ) + + self.shared_expert_gate = heuristics.instantiate_linear(linear_config, self._engine_config) + + def make_norm_layer(self) -> None: + """ + Instantiates the normalization layer for the model. This sets the `self.norm` attribute. + + TODO(cmikeh2): In the future we'll distinguish between the different norm objects, + but for now we'll just use the same one for all of them. + """ + norm_config = DSNormConfig( + max_tokens=self._engine_config.state_manager.max_ragged_batch_size, + type=self.norm_type, + channels=self.model_dim, + residual_dtype=self.activation_dtype, + input_dtype=self.activation_dtype, + output_dtype=self.activation_dtype, + eps=self._config.rms_norm_eps, + ) + + self.norm = heuristics.instantiate_pre_norm(norm_config, self._engine_config) + + """ + Model implementation + """ + + def __init__(self, config: DSModelImplementationConfig, engine_config: RaggedInferenceEngineConfig, + base_mp_group: MPType) -> None: + """ + Base implementation for initialization. By default, this will initialize + the traditional components of a transformer model: + - Embedding + - QKV projection + - Self attention + - Attention output projection + - Feed forward network + - Normalization + - Unembedding + + Arguments: + config (DSModelImplementationConfig): Model-specific configuration. No assumptions + should be made about this config that are not closely tied to the specific + model implementation. + engine_config (RaggedInferenceEngineConfig): Engine configuration. + base_mp_group (MPType): Base communication group for Tensor-parallel inference. + """ + super().__init__(config, engine_config, base_mp_group) + + self.make_norm_layer() + self.make_qkv_layer() + self.make_attn_layer() + self.make_attn_out_layer() + self.make_moe_layer() + self.make_shared_expert_mlp_1_layer() + self.make_shared_expert_mlp_2_layer() + self.make_shared_expert_gate_layer() + self.make_embedding_layer() + self.make_unembedding_layer() + self._kv_cache_config = None + + """ + Forward implementations + """ + + def _forward_embed(self, ragged_batch: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs the embedding lookup prior to running the transformer of the model. + + Arguments: + ragged_batch (RaggedBatchWrapper): The batch to embed. + + Returns: + torch.Tensor: The embedded batch. + """ + embed = self.embed(ragged_batch, self._non_transformer.word_emb) + + if embed.shape[-1] != self.model_dim: + raise ValueError(f"Embedding output shape {embed.shape} does not match model_dim {self.model_dim}") + + return embed + + def _forward_transformer(self, layer_idx: int, residual: torch.Tensor, hidden_states: torch.Tensor, + ragged_batch_info: RaggedBatchWrapper) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Executes one (slightly offset) layer of the transformer. This implementation does a peak-ahead + optimization to fuse the layer norm of the next layer into the current layer. + + Arguments: + layer_idx (int): The index of the layer to execute. + residual (torch.Tensor): The residual tensor from the previous layer. + hidden_states (torch.Tensor): The hidden states from the previous layer. This is the + hidden states after pre normalization. + ragged_batch_info (RaggedBatchWrapper): The batch metadata. + """ + # TODO(cmikeh2): Distribute ragged_batch_info to all modules + + cur_params = self._transformer[layer_idx] + kv_cache = self.state_manager.get_cache(layer_idx) + + hidden_states = self.qkv(hidden_states, cur_params.qkv_w, b=cur_params.qkv_b) + hidden_states = self.attn(hidden_states, kv_cache, ragged_batch_info) + hidden_states = self.attn_out(hidden_states, cur_params.attn_out_w, b=None) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + residual, hidden_states = self.norm(residual, hidden_states, cur_params.mlp_norm_gamma, beta=None) + + shared_expert_output = self.shared_expert_mlp_1(hidden_states, cur_params.shared_moe_mlp_1, b=None) + shared_expert_output = self.shared_expert_mlp_2(shared_expert_output, cur_params.shared_moe_mlp_2, b=None) + shared_expert_gate_output = self.shared_expert_gate(hidden_states, cur_params.shared_moe_gate, b=None)[..., :1] + # shared_expert_gate_output shape[-1] is 1 + shared_expert_output.mul_(torch.sigmoid(shared_expert_gate_output)) + hidden_states = self.moe(hidden_states, ragged_batch_info, cur_params.moe_gate, cur_params.moe_mlp_1, + cur_params.moe_mlp_2) + hidden_states.add_(shared_expert_output) + + if self.tp_size > 1: + dist.all_reduce(hidden_states, group=self._base_mp_group) + + if layer_idx != self.num_layers - 1: + next_params = self._transformer[layer_idx + 1] + residual, hidden_states = self.norm(residual, hidden_states, next_params.attn_norm_gamma, beta=None) + else: + # On last layer, we just need to perform the residual add. Adding into the residual + # here is safe. + residual.add_(hidden_states) + + return residual, hidden_states + + def _forward_unembed(self, hidden_states: torch.Tensor, ragged_batch_info: RaggedBatchWrapper) -> torch.Tensor: + """ + Performs unembedding of the hidden states to logits. This will only sample the final + token of each sequence. + """ + logits = self.unembed(hidden_states, + self._non_transformer.word_unembed, + ragged_batch_info, + gamma=self._non_transformer.final_norm) + + if self.tp_size > 1: + comm_buffer = empty_from(self._comm_logits, (self.tp_size, logits.shape[0], logits.shape[1])) + full_logits = empty_from(self._return_logits, (logits.shape[0], self.vocab_size)) + + dist.all_gather_into_tensor(comm_buffer, logits, group=self._base_mp_group) + + full_logits.copy_(comm_buffer.permute(1, 0, 2).reshape(logits.shape[0], self.vocab_size)) + + return full_logits + else: + return logits + + def forward(self, wrapped_batch: RaggedBatchWrapper) -> torch.Tensor: + + residual = self._forward_embed(wrapped_batch) + + residual, hidden_states = self.norm(residual, None, self._transformer[0].attn_norm_gamma, beta=None) + + for layer_idx in range(self.num_layers): + residual, hidden_states = self._forward_transformer(layer_idx, residual, hidden_states, wrapped_batch) + + return self._forward_unembed(residual, wrapped_batch) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/policy.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/policy.py new file mode 100644 index 0000000000000000000000000000000000000000..630bafe993a886cbd9d6e83b58450b711c870319 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/qwen_v2_moe/policy.py @@ -0,0 +1,30 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any + +from ...config_v2 import RaggedInferenceEngineConfig +from ..inference_policy_base import ContainerMap, InferenceV2Policy +from .container import Qwen2MoeNonTransformerContainer, Qwen2MoeTransformerContainer +from .model import Qwen2MoeInferenceModel + + +class Qwen2MoePolicy(InferenceV2Policy): + + def instantiate_model(self, engine_config: RaggedInferenceEngineConfig, mp_group: Any) -> Qwen2MoeInferenceModel: + return Qwen2MoeInferenceModel(config=self._model_config, engine_config=engine_config, base_mp_group=mp_group) + + def build_container_map(self) -> ContainerMap: + map = ContainerMap() + + transformer_containers = [Qwen2MoeTransformerContainer(self.model) for _ in range(self.model.num_layers)] + + map.set_transformer_params(['model.layers'], transformer_containers) + + map.set_non_transformer_params(Qwen2MoeNonTransformerContainer(self.model)) + + map.set_unmapped_params([]) + + return map diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..63421bc1c622822f3975680bdbbcffe66b261c4d --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__init__.py @@ -0,0 +1,12 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .attn import * +from .attn_out import * +from .embedding import * +from .mlp import * +from .qkv import * +from .types import * +from .unembed import * diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..87fdc72d69c9ff8faac1c5afeafd4a51e2eef73c Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/attn.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/attn.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f7bcad2c69c07e83e9de9726e99e4fb71badc9e4 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/attn.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/attn_out.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/attn_out.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b9548042e88840fab8afa21c623eb738d6d7b8de Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/attn_out.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/embedding.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/embedding.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f7e44c5790ec870768857d1654f2d85493bbb4a7 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/embedding.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/mlp.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/mlp.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..1512768a69310c173849c677a77f29f8deeea05e Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/mlp.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/qkv.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/qkv.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..996ae8af37e102b57dbe5dc125a3b8ed78be5117 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/qkv.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/types.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/types.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..28134bb42d5248f277465cc8a439cc88773961a7 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/types.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/unembed.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/unembed.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..4c8237ecce14585c89089cdbdabbbda144308c93 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/unembed.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/utils.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/utils.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..9561fa350d3d328fdba093d10496dbc0ff094fee Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/__pycache__/utils.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/attn.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/attn.py new file mode 100644 index 0000000000000000000000000000000000000000..de8d6f6ac4c56dc591ea1dce59ef3b7e4f8b2b8b --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/attn.py @@ -0,0 +1,63 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Optional, Tuple + + +def get_local_heads(shard_rank: int, + num_shards: int, + n_heads_q: int, + n_heads_kv: Optional[int] = None) -> Tuple[int, int]: + """ + Helper to determine the number of local heads of a given shard. + + Args: + shard_rank (int): The rank of the shard. + num_shards (int): The total number of shards that attention is distributed over. + n_heads_q (int): The number of query heads. + n_heads_kv (int): The number of key/value heads. If not passed, it is assumed that + the number of query and key/value heads are the same. + """ + if n_heads_q < num_shards: + raise ValueError("There must be at least as many attention heads as there are shards.") + + if n_heads_kv is None or n_heads_kv == n_heads_q: + # MHA attention + base_heads = n_heads_q // num_shards + extra_heads = n_heads_q % num_shards + + if shard_rank < extra_heads: + return (base_heads + 1), (base_heads + 1) + else: + return base_heads, base_heads + else: + # GQA attention + if n_heads_q % n_heads_kv != 0: + raise ValueError("Must be an even ratio between query and key/value heads.") + + if n_heads_kv < num_shards and num_shards % n_heads_kv != 0: + raise ValueError( + "If splitting a group across multiple shards, we must be able to distribute the groups evenly.") + + if n_heads_kv >= num_shards and n_heads_kv % num_shards != 0: + raise ValueError("If parallelizing groups, must be able to evenly distribute them.") + + q_ratio = n_heads_q // n_heads_kv + + if n_heads_kv >= num_shards: + local_kv_heads = n_heads_kv // num_shards + local_q_heads = local_kv_heads * q_ratio + return local_q_heads, local_kv_heads + else: + group_sharding_size = num_shards // n_heads_kv + group_rank_idx = shard_rank % group_sharding_size + + base_heads = q_ratio // group_sharding_size + extra_heads = q_ratio % group_sharding_size + + if group_rank_idx < extra_heads: + return (base_heads + 1), 1 + else: + return base_heads, 1 diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/attn_out.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/attn_out.py new file mode 100644 index 0000000000000000000000000000000000000000..ce7c105531eaba2c3948b385282b2bd1c3358b5f --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/attn_out.py @@ -0,0 +1,111 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Optional + +import torch + +from .types import ShardingType +from .utils import shard_param, get_shard_endpoints + + +def shard_attn_out_param(param: torch.Tensor, + shard_rank: int, + num_shards: int, + head_size: int, + n_heads_q: Optional[int] = None, + n_heads_kv: Optional[int] = None) -> Optional[torch.Tensor]: + """ + Utility method for sharding an attention output parameter. + """ + if len(param.shape) == 1: + # We will do the bias addition on the 0th rank only rather than scale the parameter and + # implicitly reconstruct this in the distributed reduce. + return param if shard_rank == 0 else None + + assert n_heads_kv is None or (n_heads_q is not None + and n_heads_kv is not None), "n_heads_kv should not be passed without n_heads_q" + + mha_sharding = n_heads_kv is None or n_heads_q == n_heads_kv + + if mha_sharding: + return shard_param(param, ShardingType.INNER_DIMENSION, shard_rank, num_shards, granularity=head_size) + else: + assert param.shape[0] == head_size * n_heads_q, "GQA param shape is not correct" + + # 32 KV heads, 16 shards for example + even_kv_sharding = n_heads_kv % num_shards == 0 + + # 8 KV heads, 16 shards for example + even_kv_distribution = num_shards % n_heads_kv == 0 + + assert even_kv_sharding or even_kv_distribution, "No partitioning algorithm for this yet." + + if even_kv_sharding: + # Same as original sharding scenario + return shard_param(param, ShardingType.INNER_DIMENSION, shard_rank, num_shards, granularity=head_size) + else: + # We will first do a sharding on the KV and Q to map to the one KV shard per group of Q. + q_sharding_degree = num_shards // n_heads_kv + + kv_head = shard_rank // q_sharding_degree + + q_sharding_rank = shard_rank % q_sharding_degree + q_factor = n_heads_q // n_heads_kv + + q_chunk = param[..., q_factor * kv_head * head_size:q_factor * (kv_head + 1) * head_size] + + return shard_param(q_chunk, + ShardingType.INNER_DIMENSION, + q_sharding_rank, + q_sharding_degree, + granularity=head_size) + + +def attn_out_in_features(out_features: int, + shard_rank: int, + num_shards: int, + head_size: int, + n_heads_q: Optional[int] = None, + n_heads_kv: Optional[int] = None) -> int: + """ + Helper to calculate the expected output projection dimension of a QKV projection matrix. + + Args: + in_features (int): The model dimension. + shard_rank (int): Which rank to return the corresponding size for. + num_shards (int): The total number of shards the parameter is distributed across. + head_size (int): The size of each attention head. + n_heads_q (int): The number of query heads on the model. This only needs to be passed if the number + of query and key/value heads are different. If passed without n_heads_kv, default + MHA partitioning will be used. + n_heads_kv (int): The number of key and value heads on the model. This only needs to be passed + if the number of query and key/value heads are different. This argument cannot be passed without + also passing n_heads_q (we want to explicitly opt into GQA sharding). + """ + assert n_heads_kv is None or (n_heads_q is not None + and n_heads_kv is not None), "n_heads_kv should not be passed without n_heads_q" + + mha_sharding = n_heads_kv is None or n_heads_q == n_heads_kv + + if mha_sharding: + endpoints = get_shard_endpoints(out_features, shard_rank, num_shards, granularity=head_size) + return endpoints[1] - endpoints[0] + else: + if n_heads_kv >= num_shards: + assert n_heads_kv % num_shards == 0, "No partitioning algorithm for this yet." + n_local_groups = n_heads_kv // num_shards + group_size = n_heads_q // n_heads_kv + + return n_local_groups * head_size * group_size + else: + assert num_shards % n_heads_kv == 0, "No partitioning algorithm for this yet." + q_split_degree = num_shards // n_heads_kv + q_split_rank = shard_rank % q_split_degree + split_granularity = (n_heads_q // n_heads_kv) * head_size + + q_endpoints = get_shard_endpoints(split_granularity, q_split_rank, q_split_degree, granularity=head_size) + + return q_endpoints[1] - q_endpoints[0] diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/embedding.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/embedding.py new file mode 100644 index 0000000000000000000000000000000000000000..00d335768ae69b1d897dac92c4c222fdedd1e609 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/embedding.py @@ -0,0 +1,34 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from .types import ShardingType +from .utils import shard_param, get_shard_endpoints + + +def shard_embedding_param(param: torch.Tensor, shard_rank: int, num_shards: int) -> torch.Tensor: + """ + Utility method for sharding an embedding parameter. + + Args: + param (torch.Tensor): The parameter to shard. Should be of shape [vocab_size, model_dim] + shard_rank (int): Which shard of the partitioned tensor to return. + num_shards (int): The total number of shards the parameter is distributed across. + """ + return shard_param(param, ShardingType.INNER_DIMENSION, shard_rank, num_shards) + + +def sharded_embedding_dim(embedding_size: int, shard_rank: int, num_shards: int) -> int: + """ + Utility method for getting the size of the embedding dimension of a sharded embedding. + + Args: + embedding_size (int): The size of the embedding. + shard_rank (int): Which shard of the partitioned tensor to return. + num_shards (int): The total number of shards the parameter is distributed across. + """ + start_idx, end_idx = get_shard_endpoints(embedding_size, shard_rank, num_shards) + return end_idx - start_idx diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/mlp.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/mlp.py new file mode 100644 index 0000000000000000000000000000000000000000..8abd0ff8622df3bd302f13292b48e9be9cae9782 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/mlp.py @@ -0,0 +1,75 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Optional + +import torch + +from .types import ShardingType, DEFAULT_SHARD_GRANULARITY +from .utils import shard_param, get_shard_endpoints + + +def shard_mlp_1_param(param: torch.Tensor, + shard_rank: int, + num_shards: int, + gated: bool = False, + is_moe: bool = False) -> torch.Tensor: + """ + Utility method for sharding an MLP 1 parameter. Both biases and weights are supported, as well + as for fused weights for MoE. + + Args: + param (torch.Tensor): The parameter to shard. + shard_rank (int): Which shard of the partitioned tensor to return. + num_shards (int): The total number of shards the parameter is distributed across. + gated (bool): Whether or not the parameter is from a gated MLP. + """ + bias_dims = 2 if is_moe else 1 + + if gated: + return shard_param(param, + ShardingType.OUTER_DIMENSION, + shard_rank, + num_shards, + granularity=DEFAULT_SHARD_GRANULARITY * 2, + bias_dims=bias_dims) + else: + return shard_param(param, ShardingType.OUTER_DIMENSION, shard_rank, num_shards, bias_dims=bias_dims) + + +def shard_mlp_2_param(param: torch.Tensor, + shard_rank: int, + num_shards: int, + is_moe: bool = False) -> Optional[torch.Tensor]: + """ + Utility method for sharding an MLP 2 parameter. + + Args: + param (torch.Tensor): The parameter to shard. + shard_rank (int): Which shard of the partitioned tensor to return. + num_shards (int): The total number of shards the parameter is distributed across. + is_moe (bool): Whether or not the parameter is from a MoE model. + """ + bias_dim_size = 2 if is_moe else 1 + + if len(param.shape) == bias_dim_size: + # We will do the bias addition on the 0th rank only rather than scale the parameter and + # implicitly reconstruct this in the distributed reduce. + return param if shard_rank == 0 else None + + return shard_param(param, ShardingType.INNER_DIMENSION, shard_rank, num_shards) + + +def sharded_intermediate_dim(intermediate_size: int, num_shards: int, shard_rank: int) -> int: + """ + Utility method for getting the size of the intermediate dimension of a sharded MLP. + + Args: + intermediate_size (int): The size of the intermediate dimension. + num_shards (int): The total number of shards the parameter is distributed across. + shard_rank (int): Which shard of the partitioned tensor to return. + """ + endpoints = get_shard_endpoints(intermediate_size, shard_rank, num_shards) + return endpoints[1] - endpoints[0] diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/qkv.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/qkv.py new file mode 100644 index 0000000000000000000000000000000000000000..2b6d7f40836e8ac72d479b2c2792551e38e3d379 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/qkv.py @@ -0,0 +1,166 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Optional + +import torch + +from .types import ShardingType +from .utils import shard_param, get_shard_endpoints + + +def shard_qkv_param(param: torch.Tensor, + shard_rank: int, + num_shards: int, + head_size: int, + n_heads_q: Optional[int] = None, + n_heads_kv: Optional[int] = None) -> Optional[torch.Tensor]: + """ + Utility method for sharding a QKV parameter. Both biases and weights are supported. It is assumed + that the layout of the parameter is such that all Q heads, all K heads, and all V heads + are contiguous with respect to each other. + + Args: + param (torch.Tensor): The parameter to shard. + shard_rank (int): Which shard of the partitioned tensor to return. + num_shards (int): The total number of shards the parameter is distributed across. + head_size (int): The size of each head. + n_heads_q (int): The number of query heads. This only needs to be passed if the number + of query and key/value heads are different. If passed without n_heads_kv, default + MHA partitioning will be used. + n_heads_kv (int): The number of key/value heads. This only needs to be passed if the number + of query and key/value heads are different. This argument should not be passed without + n_heads_q (we want to explicitly opt into GQA sharding). + """ + if n_heads_kv is not None and n_heads_q is None: + raise ValueError("n_heads_kv should not be passed without n_heads_q") + + if n_heads_q is None: + # Guaranteed to be in MHA + if param.shape[0] // 3 % head_size != 0: + raise ValueError("MHA param shape is not correct") + n_heads_q = param.shape[0] // head_size // 3 + mha_sharding = True + else: + mha_sharding = n_heads_q == n_heads_kv + + if n_heads_q < num_shards: + raise ValueError("There must be at least as many query heads as there are shards.") + + if mha_sharding: + return shard_param(param, + ShardingType.OUTER_DIMENSION, + shard_rank, + num_shards, + num_concatenated_matrices=3, + granularity=head_size) + else: + if n_heads_q % n_heads_kv != 0: + raise ValueError("Must be an even ratio between query and key/value heads.") + + if param.shape[0] != head_size * (n_heads_q + 2 * n_heads_kv): + raise ValueError("GQA param shape is not correct") + + # 32 KV heads, 16 shards for example + if n_heads_kv >= num_shards and n_heads_kv % num_shards != 0: + raise ValueError("Currently do not support uneven partitioning of KV heads for GQA.") + + # 8 KV heads, 16 shards for example + if n_heads_kv < num_shards and num_shards % n_heads_kv != 0: + raise ValueError("Currently do not support distributing KV heads across different numbers of shards.") + else: + even_kv_sharding = n_heads_kv >= num_shards + + if param is None: + return None + + q_param = param[:head_size * n_heads_q] + kv_param = param[head_size * n_heads_q:] + + if even_kv_sharding: + # This is equivalent to the original sharding algorithm since n_heads_q = C * n_heads_kv. + # If n_heads_kv % num_shards == 0, then n_heads_q % num_shards == 0. + q_param = shard_param(q_param, ShardingType.OUTER_DIMENSION, shard_rank, num_shards, granularity=head_size) + kv_param = shard_param(kv_param, + ShardingType.OUTER_DIMENSION, + shard_rank, + num_shards, + num_concatenated_matrices=2, + granularity=head_size) + return torch.cat([q_param, kv_param], dim=0) + else: + # We will first do a sharding on the KV and Q to map to the one KV shard per group of Q. + q_sharding_degree = num_shards // n_heads_kv + + kv_head = shard_rank // q_sharding_degree + k_param = kv_param[kv_head * head_size:(kv_head + 1) * head_size] + v_param = kv_param[(n_heads_kv + kv_head) * head_size:(n_heads_kv + kv_head + 1) * head_size] + + q_sharding_rank = shard_rank % q_sharding_degree + q_factor = n_heads_q // n_heads_kv + + q_chunk = q_param[q_factor * kv_head * head_size:q_factor * (kv_head + 1) * head_size] + + q_param = shard_param(q_chunk, + ShardingType.OUTER_DIMENSION, + q_sharding_rank, + q_sharding_degree, + granularity=head_size) + + return torch.cat([q_param, k_param, v_param], dim=0) + + +def qkv_out_features(in_features: int, + shard_rank: int, + num_shards: int, + head_size: int, + n_heads_q: Optional[int] = None, + n_heads_kv: Optional[int] = None) -> int: + """ + Helper to calculate the expected output projection dimension of a QKV projection matrix. + + Args: + in_features (int): The model dimension. + shard_rank (int): Which rank to return the corresponding size for. + num_shards (int): The total number of shards the parameter is distributed across. + head_size (int): The size of each head. + n_heads_q (int): The number of query heads. This only needs to be passed if the number + of query and key/value heads are different. If passed without n_heads_kv, default + MHA partitioning will be used. + n_heads_kv (int): The number of key/value heads. This only needs to be passed if the number + of query and key/value heads are different. This argument cannot be passed without also + passing n_heads_q (we want to explicitly opt into GQA sharding). + """ + if n_heads_kv is not None and n_heads_q is None: + raise ValueError("n_heads_kv should not be passed without n_heads_q") + + mha_sharding = n_heads_kv is None or n_heads_q == n_heads_kv + + if n_heads_q is not None and in_features != head_size * n_heads_q: + raise ValueError("in_features is not consistent with n_heads_q and head_size") + + if mha_sharding: + endpoints = get_shard_endpoints(in_features, shard_rank, num_shards, granularity=head_size) + return (endpoints[1] - endpoints[0]) * 3 + else: + if n_heads_kv >= num_shards: + if n_heads_kv % num_shards != 0: + raise ValueError("The KV heads must be evenly distributed across the shards.") + + n_local_groups = n_heads_kv // num_shards + group_size = n_heads_q // n_heads_kv + + return n_local_groups * head_size * (2 + group_size) + else: + if num_shards % n_heads_kv != 0: + raise ValueError("A shared KV head must always partition across the same number of shards.") + + q_split_degree = num_shards // n_heads_kv + q_split_rank = shard_rank % q_split_degree + split_granularity = (n_heads_q // n_heads_kv) * head_size + + q_endpoints = get_shard_endpoints(split_granularity, q_split_rank, q_split_degree, granularity=head_size) + + return (q_endpoints[1] - q_endpoints[0]) + 2 * head_size diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/types.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/types.py new file mode 100644 index 0000000000000000000000000000000000000000..01dce0db523a9e904052c8336fd14278c5162b2c --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/types.py @@ -0,0 +1,18 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from enum import Enum + +DEFAULT_SHARD_GRANULARITY = 32 + + +class ShardingType(Enum): + # Inner dimension sharding corresponds to splitting the Tensor along the K-dimension + # of a matrix multiplication. This would be used for attention_output or MLP2. + INNER_DIMENSION = 1 + + # Outer dimension sharding corresponds to splitting the Tensor along the N-dimension + # of a matrix multiplication. This would be used for the QKV and MLP1 projections. + OUTER_DIMENSION = 0 diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/unembed.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/unembed.py new file mode 100644 index 0000000000000000000000000000000000000000..6cc771969ad9e195f04216e1374ea0ced6fb7065 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/unembed.py @@ -0,0 +1,41 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch + +from .types import ShardingType +from .utils import shard_param, get_shard_endpoints + + +def shard_unembed_param(param: torch.Tensor, shard_rank: int, num_shards: int) -> torch.Tensor: + """ + Utility method for sharding an unembed parameter. We shard unembeddings on the vocab dimension + with the expectation of an all-gather to produce the full results. + + TODO(cmikeh2): Really ideal would be if MII could have access to the comm and we would do + an A2A and sharded sampling. + + Args: + param (torch.Tensor): The parameter to shard. Should be of shape [vocab_size, model_dim] + shard_rank (int): Which shard of the partitioned tensor to return. + num_shards (int): The total number of shards the parameter is distributed across. + + Returns: + torch.Tensor: The sharded parameter of shape [sharded_vocab_size, model_dim] + """ + return shard_param(param, ShardingType.OUTER_DIMENSION, shard_rank, num_shards, granularity=1) + + +def sharded_unembed_dim(vocab_size: int, shard_rank: int, num_shards: int) -> int: + """ + Utility method for determining the sharded vocab size of a sharded unembed parameter. + + Args: + vocab_size (int): The size of the vocabulary. + shard_rank (int): Which shard of the partitioned tensor to return. + num_shards (int): The total number of shards the parameter is distributed across. + """ + start_idx, end_idx = get_shard_endpoints(vocab_size, shard_rank, num_shards, granularity=1) + return end_idx - start_idx diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/utils.py b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fd0eb51873f83834f99bf14b084cf62efbde05f3 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/model_implementations/sharding/utils.py @@ -0,0 +1,104 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Optional, Tuple + +import torch + +from .types import ShardingType, DEFAULT_SHARD_GRANULARITY + + +def get_shard_endpoints(dim_size: int, + shard_rank: int, + num_shards: int, + granularity: int = DEFAULT_SHARD_GRANULARITY) -> Tuple[int, int]: + """ + Given a dimension to shard with size dim_size, return the start and end indices of the slice + that belong to the given rank. + + The typical use of this is as an internal helper function, so see if there is a higher level + API that better suits the application. + + Args: + dim_size (int): The size of the dimension to shard. + shard_rank (int): The rank of the shard to return. + num_shards (int): Total number of shards the dimension will be distributed across. + granularity (int): The minimum alignment of the shard endpoints. This is used to support + non-even head counts as well as align dimensions to cleaner GEMM boundaries. + """ + assert dim_size % granularity == 0, "Dimension size must be divisible by granularity" + + total_chunks = dim_size // granularity + base_chunks_per_rank = total_chunks // num_shards + remainder_chunks = total_chunks % num_shards + + start_chunk_id = shard_rank * base_chunks_per_rank + min(shard_rank, remainder_chunks) + end_chunk_id = start_chunk_id + base_chunks_per_rank + (1 if shard_rank < remainder_chunks else 0) + + return start_chunk_id * granularity, end_chunk_id * granularity + + +def shard_param(param: Optional[torch.Tensor], + shard_mode: ShardingType, + shard_rank: int, + num_shards: int, + num_concatenated_matrices: int = 1, + granularity: int = 32, + bias_dims: int = 1) -> torch.Tensor: + """ + Utility for sharding a parameter. This will return the slice of the parameter that should + exist on the given shard_rank given the sharding configuration. The workflow here is + to find the minimum bounded Tensor to shard, get the slicing endpoints, and then concatenate + as needed. + + The typical use of this is as an internal helper function, so see if there is a higher level + API that better suits the application. + + Args: + param (torch.Tensor): The parameter to shard. + shard_mode (ShardingType): The type of sharding to apply. See ShardingType for more context. + shard_rank (int): The rank of the shard to return. + num_shards (int): Total number of shards the parameter will be distrbuted across. + num_concatenated_matrices (int): The number of matrices that have been concatenated together in the original + parameter. An example of this is a fused QKV projection matrix, where the `num_concatenated_matrices` + argument would be 3. + granularity (int): The minimum alignment of the shard endpoints. For attention projection matrices, this + should be set to the head size to support non-even sharding. + bias_dims (int): The number of dimensions that are considered bias dimensions. This is used to support + sharding of MoE and non-MoE biases on the same codepath. + """ + assert shard_rank < num_shards, "Shard rank must be less than num_shards" + + # Easier to hide this inside of the sharding logic than to add checks in every model + # implementation. + if param is None: + return None + + if num_shards == 1: + # Trivial case of no sharding. + return param + + if shard_mode == ShardingType.OUTER_DIMENSION: + + def get_matrices(dim_idx: int) -> torch.Tensor: + dim_size = param.size(dim_idx) // num_concatenated_matrices + start_channel_id, end_channel_id = get_shard_endpoints(dim_size, shard_rank, num_shards, granularity) + return torch.chunk(param, num_concatenated_matrices, dim=dim_idx), start_channel_id, end_channel_id + + if param.ndim == bias_dims: + # Special case for bias parameters. + matrices, start_channel_id, end_channel_id = get_matrices(dim_idx=-1) + return torch.cat([mat[..., start_channel_id:end_channel_id] for mat in matrices], dim=-1) + else: + # General case for weight parameters. This assumes MoE parameters are stored in the format of + # [num_experts, out_features, in_features] + matrices, start_channel_id, end_channel_id = get_matrices(dim_idx=-2) + return torch.cat([mat[..., start_channel_id:end_channel_id, :] for mat in matrices], dim=-2) + + elif shard_mode == ShardingType.INNER_DIMENSION: + dim_size = param.size(-1) // num_concatenated_matrices + start_channel_id, end_channel_id = get_shard_endpoints(dim_size, shard_rank, num_shards, granularity) + matrices = torch.chunk(param, num_concatenated_matrices, dim=-1) + return torch.cat([mat[..., start_channel_id:end_channel_id] for mat in matrices], dim=-1) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..917c1599de2e6124ef7ee09f73a655053f6acb7e --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__init__.py @@ -0,0 +1,8 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from . import implementations +from . import interfaces +from .module_registry import ConfigBundle diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..db2219c57b2fd2aeb47e86ec11a43f9232ec5968 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/ds_module.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/ds_module.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2081f40611dae4e0e894818fb812d3fd450f8c0e Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/ds_module.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/heuristics.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/heuristics.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d6002ae0ade58591f5051e59af9f0912b2fde964 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/heuristics.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/module_registry.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/module_registry.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ceecd69f915a760ddb40cdcbedbb025b96ee6cfa Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/__pycache__/module_registry.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..3429e69b47de32cd14346b8fd4ad1c7d0ac460c3 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__init__.py @@ -0,0 +1,16 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .attention_configs import ( + DSSelfAttentionConfig, + PositionalEmbeddingType, + MaskingType, + RotateHalfConfig, +) +from .embedding_config import DSEmbeddingsConfig +from .linear_config import DSLinearConfig +from .moe_config import DSMoEConfig +from .norm_config import DSNormConfig, NormTypeEnum +from .unembed_config import DSUnembedConfig diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f74c825f389f10c67e481629e29a86f9dd96db0d Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/attention_configs.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/attention_configs.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a9a3fe0ab6516665c32108e278d3957e358c9228 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/attention_configs.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/embedding_config.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/embedding_config.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..37b01c8ef927f55d37eb29b3da350697d3b857dd Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/embedding_config.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/linear_config.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/linear_config.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2a7f6cccb3351f04f44cd63aa030626f3d86428b Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/linear_config.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/moe_config.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/moe_config.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..eeb850f2b51e6eb415243a20c1a29a89873982e1 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/moe_config.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/norm_config.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/norm_config.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..27e832d369f5b590558f90caff2685a125b47acd Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/norm_config.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/unembed_config.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/unembed_config.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c9b4b555077f955f0e28139735462471b8a1b382 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/__pycache__/unembed_config.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/attention_configs.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/attention_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..be6a3535024c1e2e90d89d57dca9efb454b8889f --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/attention_configs.py @@ -0,0 +1,110 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from enum import Enum +from typing import Dict, Optional + +from ...inference_utils import DtypeEnum +from ...modules.ds_module import DSModuleConfig +from deepspeed.runtime.config_utils import DeepSpeedConfigModel + + +class PositionalEmbeddingType(Enum): + + # No positional embeddings + none = "none" + + # Rotary positional embeddings - every half + rotate_half = "rotate_half" + + # Rotary positional embeddings - every other + rotate_every_other = "rotate_every_other" + + # Alibi + alibi = "alibi" + + +class RotateHalfConfig(DeepSpeedConfigModel): + + use_trained_freqs: bool = False + """ + Whether to use a passed `trained_freqs` tensor for the attention implementation + or to use default synthesized frequencies. + """ + + theta_base: float = 10_000.0 + """ + Base for theta. This will only be used if `use_trained_freqs` is False. + """ + + rotate_dim: Optional[int] = None + """ + How many neurons to rotate. If None, then all neurons will be rotated. Many external configs + will set this number to half the head dimension and then internally multiply by 2. To make it + more clear to understand what is happening (rotate_dim < head_dim -> then only partial rotation), + we do not do this multiplication internally. + """ + + +class MaskingType(Enum): + + # No masking + none = "none" + + # Causal masking + causal = "causal" + + # Local masking + local = "local" + + # Symmetric masking (this is a 1D tensor mask) + symmetric = "symmetric" + + # Arbitrary masking (this would correspond to a 2D tensor mask) + asymmetric = "asymmetric" + + +class DSSelfAttentionConfig(DSModuleConfig): + """ + Config class for attention. + """ + + # Number of query attention heads on this shard + n_heads_q: int + + # Number of KV attention heads on this shard + n_heads_kv: int + + # Size of each attention head + head_size: int + + # Max number of sequences that may compose a ragged batch + max_sequences: int + + # Scale factor for attention scores + scale_factor: float = 1.0 + + # Input data type + input_dtype: DtypeEnum = DtypeEnum.fp16 + + # Output data type + output_dtype: DtypeEnum = DtypeEnum.fp16 + + # Masking type + masking_type: MaskingType = MaskingType.causal + + # Masking args + masking_args: Dict = {} + + # Positional embedding type + positional_embedding_type: PositionalEmbeddingType = PositionalEmbeddingType.none + + # Positional embedding args + positional_embedding_config: Optional[RotateHalfConfig] = None + """ + To extend this for the other positional embedding types, we would need to add + new configs for each type (as necessary) and annotate this with the + Union[RotateHalfConfig, OtherConfig, ...] type. + """ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/embedding_config.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/embedding_config.py new file mode 100644 index 0000000000000000000000000000000000000000..2486c5986e9531c7377f1dddbe81ebafb6ef377f --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/embedding_config.py @@ -0,0 +1,70 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Optional + +from ...inference_utils import DtypeEnum, NormTypeEnum +from ...modules.ds_module import DSModuleConfig +""" +Trying to define the space we need to support here right now: + +Types of embeddings I've found so far: + 1. Token embedding + 2. Position embedding + 3. Token type embedding + 4. LN + +GPTNeo: 1, 2, 3 (shared with 1) +GPTNeoX: 1 +GPTJ: 1, 3 +LLaMA: 1 +BERT: 1, 2, 3, 4 +GPT2: 1, 2, 3 (shared with 1) + +Sidebar for OPT: +OPT: 1, 2 +1 may not actually project to the actual hidden dimension according to the raw +code, but for the model configs we care about it does. +2 has a weird offset associated with it that the others do not. +""" + + +class DSEmbeddingsConfig(DSModuleConfig): + """ + Config class for DSEmbeddings. + """ + + residual_dtype: DtypeEnum = DtypeEnum.fp16 + """ + Data type the module should use for its output. + """ + + embedding_dim: int + """ + Dimensionality of the embedding projections. + """ + + positional_embedding: bool = False + """ + Whether the module should expect a positional embedding matrix. The shape of this + matrix should be of shape [max_seq_len + positional_offset, embedding_dim] + """ + + positional_offset: int = 0 + """ + Whether the linearized token IDs should be offset by a certain amount. For an example + of this, see the OPT model implementation. + """ + + use_token_type: bool = False + """ + Whether the module should expect a token type embedding matrix. + """ + + output_normalization: Optional[NormTypeEnum] = None + """ + If a the output of the embedding module should be normalized, specify here. See + ``inference.inference_utils.NormTypeEnum`` for supported values. + """ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/linear_config.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/linear_config.py new file mode 100644 index 0000000000000000000000000000000000000000..40fe0773aeeee92d505a115f85c440a753491329 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/linear_config.py @@ -0,0 +1,43 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from ...inference_utils import ActivationType, DtypeEnum +from ...modules.ds_module import DSModuleConfig + + +class DSLinearConfig(DSModuleConfig): + """ + Config class for DSLinearBase. + """ + + in_channels: int + """ + Number of input channels + """ + + out_channels: int + """ + Number of output channels. NOTE: If this linear layer is using a gated activation function, + the value for ``out_channels`` passed here should refer to the number of channels after + gating (i.e., the expected weight shape before transformations will be ``[out_channels * 2, in_channels]``). + """ + + activation: ActivationType = ActivationType.IDENTITY + """ + The activation function for this layer. See :class:`deepspeed.inference.inference_utils.ActivationType` for + supported activation functions. + """ + + input_dtype: DtypeEnum = DtypeEnum.fp16 + """ + The data type of the input tensor. See :class:`deepspeed.inference.inference_utils.DtypeEnum` for supported + data types. + """ + + output_dtype: DtypeEnum = DtypeEnum.fp16 + """ + The data type of the output tensor. See :class:`deepspeed.inference.inference_utils.DtypeEnum` for supported + data types. + """ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/moe_config.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/moe_config.py new file mode 100644 index 0000000000000000000000000000000000000000..7bc944f55e17cf4e837c2a5db1144099dd942fdd --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/moe_config.py @@ -0,0 +1,56 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from ...inference_utils import ActivationType, DtypeEnum +from ...modules.ds_module import DSModuleConfig + + +class DSMoEConfig(DSModuleConfig): + """ + Config class for DSMoEBase + """ + + model_dim: int + """ + Size of input activation. + """ + + intermediate_features: int + """ + Size of intermediate activation. Specifically, this is the number of input features + in the second linear layer. Depending on the activation function, the output of the first + linear layer may have increased dimensionality. + """ + + n_experts: int + """ + Number of experts. + """ + + top_k: int = 1 + """ + top-k gating function (like top-1 or top-2) + """ + + input_dtype: DtypeEnum = DtypeEnum.fp16 + """ + Data type for the input activations. + """ + + output_dtype: DtypeEnum = DtypeEnum.fp16 + """ + Data type for the output activations. + """ + + activation: ActivationType = ActivationType.IDENTITY + """ + Activation function of the first MLP1 + """ + + normalize_scores: bool = False + """ + Whether normalization is applied to the selected scores. If true, the module + should rescale the scores such that their sum is 1.0. + """ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/norm_config.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/norm_config.py new file mode 100644 index 0000000000000000000000000000000000000000..358982253756af4c4065d2e8cc53f7d0dd0b0287 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/norm_config.py @@ -0,0 +1,32 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from ...inference_utils import DtypeEnum, NormTypeEnum +from ...modules.ds_module import DSModuleConfig + + +class DSNormConfig(DSModuleConfig): + """ + Config class for both DSPreLN and DSPostLN. + """ + + # Type of normalization + type: NormTypeEnum + + # Number of channels in the model embedding + channels: int + + # Data type of the residual input/outputs (we assume the residual must + # be the same data type for the entire model). + residual_dtype: DtypeEnum = DtypeEnum.fp16 + + # Data type of the hidden states input + input_dtype: DtypeEnum = DtypeEnum.fp16 + + # Data type of the hidden states output + output_dtype: DtypeEnum = DtypeEnum.fp16 + + # Epsilon value for numerical stability + eps: float = 1e-5 diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/unembed_config.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/unembed_config.py new file mode 100644 index 0000000000000000000000000000000000000000..ea4cc3cc99c17af9201db4774ebc0d095539dd82 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/configs/unembed_config.py @@ -0,0 +1,39 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from ...inference_utils import DtypeEnum, NormTypeEnum +from ...modules.ds_module import DSModuleConfig +from typing import Optional + + +class DSUnembedConfig(DSModuleConfig): + """ + Config class for DSUnembed + """ + + dtype: DtypeEnum = DtypeEnum.fp16 + """ + Expected data type. + """ + + norm_type: Optional[NormTypeEnum] = None + """ + Whether the input to the unembed is normalized prior to the unembedding projection. + """ + + model_dim: int + """ + Model embedding size. + """ + + max_sequences: int + """ + Max sequences composing the ragged batch. + """ + + vocab_size: int + """ + Local vocab size (the full vocab size may have been sharded across model parallel ranks) + """ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/ds_module.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/ds_module.py new file mode 100644 index 0000000000000000000000000000000000000000..2a6d294f32666cf46ab918ba03150d3f17ae37ff --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/ds_module.py @@ -0,0 +1,62 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from abc import ABC, abstractstaticmethod +from typing import Any, Dict, Type + +import torch + +from deepspeed.runtime.config_utils import DeepSpeedConfigModel + + +class DSModuleConfig(DeepSpeedConfigModel): + + max_tokens: int + + +class DSModuleBase(torch.nn.Module, ABC): + """ + Base class for all DeepSpeed Inference modules. This class establishes + the basic attributes of a DSModule. Only abstract functionality modules should inherit + directly from this class, not specific implementations. + """ + + @abstractstaticmethod + def name() -> str: + """ + Return a memorable, human-readable name for this module. + + This will be used as a key in custom inference configurations and should only + be implemented by the children of functionality modules. + """ + ... + + @abstractstaticmethod + def config_class() -> Type[DSModuleConfig]: + """ + Return the associated config class for this module. + + This should be implemented (along with the config class) by an abstract functionality + module. + """ + ... + + @abstractstaticmethod + def supports_config(config: DSModuleConfig) -> bool: + """ + Return whether or not this module supports the given config. + + This should be implemented by the children of functionality modules and should report + whether it would be feasible to instantiate this module with the given config. + """ + ... + + def __init__(self, config: DSModuleConfig, implementation_config: Dict[str, Any] = {}) -> None: + """ + Initialize the module with the given config. + """ + super().__init__() + self._config = config + self._implementation_config = implementation_config diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/heuristics.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/heuristics.py new file mode 100644 index 0000000000000000000000000000000000000000..f719e299a4b2eb99b40c58807a14d71ac2cc2319 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/heuristics.py @@ -0,0 +1,195 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from ..config_v2 import RaggedInferenceEngineConfig +from ..inference_utils import NormTypeEnum + +from .module_registry import ConfigBundle +from ..modules.configs import ( + DSEmbeddingsConfig, + DSLinearConfig, + DSMoEConfig, + DSNormConfig, + DSSelfAttentionConfig, + DSUnembedConfig, +) +from ..modules.interfaces import ( + DSEmbeddingBase, + DSEmbeddingRegistry, + DSLinearBase, + DSLinearRegistry, + DSMoEBase, + DSMoERegistry, + DSPostNormBase, + DSPostNormRegistry, + DSPreNormBase, + DSPreNormRegistry, + DSSelfAttentionBase, + DSSelfAttentionRegistry, + DSUnembedBase, + DSUnembedRegistry, +) + + +def instantiate_attention(attention_config: DSSelfAttentionConfig, + engine_config: RaggedInferenceEngineConfig) -> DSSelfAttentionBase: + """ + Choose an appropriate attention implementation based on the given configurations. This + method is currently a stub, but as more implementations may be developed we can centralize + the logic for choosing between them here. + + Arguments: + attention_config (DSSelfAttentionConfig): Configuration for the attention module. + engine_config (RaggedInferenceEngineConfig): Configuration for the inference engine. + + Returns: + An attention module implementing the given configuration. + """ + + # Currently, we only have one implementation, so we just return it. + config = ConfigBundle(name="dense_blocked_attention", config=attention_config) + return DSSelfAttentionRegistry.instantiate_config(config) + + +def instantiate_embed(embed_config: DSEmbeddingsConfig, engine_config: RaggedInferenceEngineConfig) -> DSEmbeddingBase: + """ + Choose an appropriate embedding implementation based on the given configurations. This + method is currently a stub, but as more implementations may be developed we can centralize + the logic for choosing between them here. + + Arguments: + embed_config (DSEmbeddingsConfig): Configuration for the embedding module. + engine_config (RaggedInferenceEngineConfig): Configuration for the inference engine. + + Returns: + An embedding module implementing the given configuration. + """ + + # Currently, we only have one implementation, so we just return it. + config = ConfigBundle(name="ragged_embedding", config=embed_config) + return DSEmbeddingRegistry.instantiate_config(config) + + +def instantiate_linear(linear_config: DSLinearConfig, engine_config: RaggedInferenceEngineConfig) -> DSLinearBase: + """ + Choose an appropriate linear implementation based on the given configurations. This + method is currently a stub, but as more implementations may be developed we can centralize + the logic for choosing between them here. + + Arguments: + linear_config (DSLinearConfig): Configuration for the linear module. + engine_config (RaggedInferenceEngineConfig): Configuration for the inference engine. + + Returns: + A linear module implementing the given configuration. + """ + + quantization_mode = engine_config.quantization.quantization_mode + if quantization_mode is None: + config = ConfigBundle(name="blas_fp_linear", config=linear_config) + else: + # Currently, we only support ``quantized_wf6af16_linear`` on NVIDIA Ampere GPUs. + if quantization_mode == "wf6af16": + import torch + if not torch.cuda.is_available(): #ignore-cuda + raise ValueError("WF6AF16 quantization is only supported on CUDA") + else: + is_rocm_pytorch = hasattr(torch.version, 'hip') and torch.version.hip is not None + if is_rocm_pytorch: + raise ValueError("WF6AF16 quantization is only supported on NVIDIA GPUs") + elif torch.cuda.get_device_properties(0).major != 8: #ignore-cuda + raise ValueError("WF6AF16 quantization is only supported on Ampere architectures") + config = ConfigBundle(name="quantized_wf6af16_linear", config=linear_config) + else: + raise ValueError(f"Unsupported quantization mode: {quantization_mode}") + return DSLinearRegistry.instantiate_config(config) + + +def instantiate_moe(moe_config: DSMoEConfig, engine_config: RaggedInferenceEngineConfig) -> DSMoEBase: + """ + Choose an appropriate MoE implementation based on the given configurations. This + method is currently a stub, but as more implementations may be developed we can centralize + the logic for choosing between them here. + + Arguments: + moe_config (DSMoEConfig): Configuration for the MoE module. + engine_config (RaggedInferenceEngineConfig): Configuration for the inference engine. + + Returns: + A MoE module implementing the given configuration. + """ + + moe_type = "cutlass_multi_gemm_moe" + + if moe_type == "cutlass_multi_gemm_moe": + # TODO: Get this off an engine config + implementation_config = { + "weight_dtype": moe_config.input_dtype, + } + + # Currently, we only have one implementation, so we just return it. + config = ConfigBundle(name="cutlass_multi_gemm_moe", + config=moe_config, + implementation_config=implementation_config) + return DSMoERegistry.instantiate_config(config) + + +def instantiate_post_norm(norm_config: DSNormConfig, engine_config: RaggedInferenceEngineConfig) -> DSPostNormBase: + """ + Choose an appropriate post-norm implementation based on the given configurations. This + method is currently a stub, but as more implementations may be developed we can centralize + the logic for choosing between them here. + + Arguments: + norm_config (DSNormConfig): Configuration for the post-norm module. + engine_config (RaggedInferenceEngineConfig): Configuration for the inference engine. + + Returns: + A post-norm module implementing the given configuration. + """ + + # Currently, we only have one implementation, so we just return it. + config = ConfigBundle(name="cuda_post_ln", config=norm_config) + return DSPostNormRegistry.instantiate_config(config) + + +def instantiate_pre_norm(norm_config: DSNormConfig, engine_config: RaggedInferenceEngineConfig) -> DSPreNormBase: + """ + Choose an appropriate pre-norm implementation based on the given configurations. Currently, + this will select between two CUDA implementations, one for LayerNorm and one for RMSNorm. + + Arguments: + norm_config (DSNormConfig): Configuration for the pre-norm module. + engine_config (RaggedInferenceEngineConfig): Configuration for the inference engine. + + Returns: + A pre-norm module implementing the given configuration. + """ + if NormTypeEnum(norm_config.type) == NormTypeEnum.LayerNorm: + module_name = "cuda_pre_ln" + elif NormTypeEnum(norm_config.type) == NormTypeEnum.RMSNorm: + module_name = "cuda_pre_rms" + + config = ConfigBundle(name=module_name, config=norm_config) + return DSPreNormRegistry.instantiate_config(config) + + +def instantiate_unembed(unembed_config: DSUnembedConfig, engine_config: RaggedInferenceEngineConfig) -> DSUnembedBase: + """ + Choose an appropriate unembedding implementation based on the given configurations. This + method is currently a stub, but as more implementations may be developed we can centralize + the logic for choosing between them here. + + Arguments: + unembed_config (DSUnembedConfig): Configuration for the unembed module. + engine_config (RaggedInferenceEngineConfig): Configuration for the inference engine. + + Returns: + An unembed module implementing the given configuration. + """ + + # Currently, we only have one implementation, so we just return it. + config = ConfigBundle(name="ragged_unembed", config=unembed_config) + return DSUnembedRegistry.instantiate_config(config) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..1b500a9a0b5a25b17696bffcbcc5fa1ffd582de4 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/__init__.py @@ -0,0 +1,13 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# Imports for registering ops +from .attention import * +from .linear import * +from .post_norm import * +from .pre_norm import * +from .embedding import * +from .unembed import * +from .moe import * diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..bbaa05411189728aae8beaf7fb19804b711089d0 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..1ff593ac1368235cc03b427532154ec92b8bbe82 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .dense_blocked_attention import DSDenseBlockedAttention diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b5fc2b4c0beae50ac5aff287f5e7cd2fe3858921 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/__pycache__/dense_blocked_attention.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/__pycache__/dense_blocked_attention.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a1d6fb43e725f0ed1f8069b8b101ea0454781256 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/__pycache__/dense_blocked_attention.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/dense_blocked_attention.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/dense_blocked_attention.py new file mode 100644 index 0000000000000000000000000000000000000000..3515b3c2b690c6346db7afba02b5790a2eb98982 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/attention/dense_blocked_attention.py @@ -0,0 +1,180 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any, Dict, Optional + +import torch + +from deepspeed.accelerator import get_accelerator +from ....allocator import empty_from +from ....inference_utils import DtypeEnum +from ....kernels.ragged_ops import ( + AtomBuilder, + BlockedFlashAttn, + BlockedRotaryEmbeddings, + BlockedTrainedRotaryEmbeddings, + get_q_block_size, + get_kv_block_size, + LinearBlockedKVCopy, +) +from ....ragged import RaggedBatchWrapper, split_kv +from deepspeed.ops.op_builder import RaggedUtilsBuilder + +from ...interfaces import DSSelfAttentionBase, DSSelfAttentionRegistry +from ...configs import DSSelfAttentionConfig, PositionalEmbeddingType, MaskingType + +try: + from functools import cached_property +except ImportError: + + def cached_property(func): + return property(func) + + +@DSSelfAttentionRegistry.register_module +class DSDenseBlockedAttention(DSSelfAttentionBase): + """ + Self attention implementation for dense, blocked self attention. + """ + + @staticmethod + def name() -> str: + return 'dense_blocked_attention' + + @staticmethod + def supports_config(config: DSSelfAttentionConfig) -> bool: + + if config.input_dtype != config.output_dtype: + return False + + if DtypeEnum(config.input_dtype) not in (DtypeEnum.fp16, DtypeEnum.bf16): + return False + + if PositionalEmbeddingType(config.positional_embedding_type) not in [ + PositionalEmbeddingType.none, PositionalEmbeddingType.rotate_half + ]: + return False + + if MaskingType(config.masking_type) != MaskingType.causal: + return False + + return True + + def __init__(self, config: DSSelfAttentionConfig, implementation_config: Dict[str, Any]) -> None: + """ + Create the Attention DSModule. + + Args: + config (DSSelfAttentionConfig): The self attention config for all attention DSModules. + implementation_config (Dict[str, Any]): + There are two (dependent) potential components in the implementtion config. + + 1. `trained_freqs` - If the embedding weights for RoPE are trained, the implementation + config should contain {'trained_freqs': True}. This will mean the implementation will + expect a `trained_freqs` tensor in the `forward` method and will not synthesize the + values internally. + + 2. `theta_base` - The base value for synthesized frequencies in the rotary embeddings. + This will only be used if `trained_freqs` is False or not present in the `implementation_config`. If this is not included, the default value of 10000.0 will be used. + """ + super().__init__(config, implementation_config) + + embed_type = PositionalEmbeddingType(config.positional_embedding_type) + if embed_type == PositionalEmbeddingType.none: + self._kv_copy = LinearBlockedKVCopy(self._config.head_size, self._config.n_heads_q, + self._config.n_heads_kv, self._config.input_dtype) + elif embed_type == PositionalEmbeddingType.rotate_half: + rotary_config = config.positional_embedding_config + assert rotary_config is not None, "Rotary config must be provided if using rotate_half as Positional Embedding Type." + + if rotary_config.use_trained_freqs: + # Theta and rotary dim are effectively embedded into either the values (theta) or the shape (rotary_dim) + # of the trained_freqs tensor. + self._kv_copy = BlockedTrainedRotaryEmbeddings(self._config.head_size, self._config.n_heads_q, + self._config.n_heads_kv, self._config.input_dtype) + else: + theta_base = rotary_config.theta_base + rotary_dim = rotary_config.rotate_dim if rotary_config.rotate_dim is not None else self._config.head_size + self._kv_copy = BlockedRotaryEmbeddings(self._config.head_size, self._config.n_heads_q, + self._config.n_heads_kv, self._config.input_dtype, rotary_dim, + theta_base) + + self._softmax_scale = self._config.scale_factor + + # TODO(cmikeh2): Attention kernel gets created here. + self._attn_kernel = BlockedFlashAttn(self._config.head_size, self._config.input_dtype) + self._atom_builder = AtomBuilder() + + self.model_dim = self._config.head_size * self._config.n_heads_q + self._output = torch.empty((self._config.max_tokens, self._config.head_size * self._config.n_heads_q), + dtype=self._config.output_dtype, + device=get_accelerator().current_device()) + + # TODO(cmikeh2): Pre-allocate storage buffer for the attention atoms. + self._max_atoms = self._config.max_sequences + self._atoms = torch.empty((self._max_atoms, 8), dtype=torch.int32, device=get_accelerator().current_device()) + + alloc_func = RaggedUtilsBuilder().load().allocate_fast_host_buffer + self._atoms_shadow = alloc_func(self._atoms) + self._cur_atoms = 0 + + @cached_property + def kv_block_size(self) -> int: + """ + Return preferred granulatity for blocked KV-cache implementation. + """ + return get_kv_block_size(self._config.head_size) + + @cached_property + def q_block_size(self) -> int: + """ + Property to calculate blocking granularity for the query dimension. + This has no impact on the KV-cache structure, but will affect the + number of attention atoms associated with a batch. + """ + return get_q_block_size(self._config.head_size) + + def build_atoms(self, ragged_batch: RaggedBatchWrapper) -> None: + """ + Build the atoms for the attention kernel. + + Args: + ragged_batch (RaggedBatchWrapper): The input ids and associated ragged batch metadata. + """ + host_atoms, n_atoms = self._atom_builder(self._atoms_shadow, ragged_batch, self.q_block_size, + self.kv_block_size) + + self._cur_atoms = n_atoms + self._atoms[:n_atoms].copy_(host_atoms[:n_atoms], non_blocking=True) + + def forward(self, + q_k_v: torch.Tensor, + kv_cache: torch.Tensor, + batch: RaggedBatchWrapper, + inv_freqs: Optional[torch.Tensor] = None) -> torch.Tensor: + """ + Forward implementation. + + Args: + q_k_v (torch.Tensor): Query/Key/Value projection Tensor of shape + [n_heads, (n_heads_q + 2 * n_heads_kv) * head_size]. + kv_cache (torch.Tensor): Blocked persistent cache of shape + [2, batch, block_size, n_heads_kv, head_size]. + batch (RaggedBatchWrapper): The input ids and associated ragged batch metadata. + inv_freqs (Optional[torch.Tensor]): The inverse frequencies for the rotary embeddings if they + have been modified from synthesizable values. + """ + if inv_freqs is not None: + self._kv_copy(kv_cache, q_k_v, batch, inv_freqs) + else: + self._kv_copy(kv_cache, q_k_v, batch) + + q = q_k_v[:, :self._config.head_size * self._config.n_heads_q] + output = empty_from(self._output, q.shape) + k_cache, v_cache = split_kv(kv_cache) + + self._attn_kernel(output, q, k_cache, v_cache, self._atoms[:self._cur_atoms], self._softmax_scale) + + return output diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..5458a855abf45c3870c2208663dd5f49fb4fb22b --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .ragged_embedding import DSRaggedEmbedding diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..eaa8d4e31ea20e181d40e0e414d8520b0c477309 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/__pycache__/ragged_embedding.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/__pycache__/ragged_embedding.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..295b0ec09603f5a95eed90e6197c525cec3fe683 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/__pycache__/ragged_embedding.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/ragged_embedding.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/ragged_embedding.py new file mode 100644 index 0000000000000000000000000000000000000000..90cdd39d1be7f7e51da65a1a517325e585386a72 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/embedding/ragged_embedding.py @@ -0,0 +1,77 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any, Dict, Optional + +import torch + +from deepspeed.accelerator import get_accelerator +from ....allocator import empty_from +from ....inference_utils import DtypeEnum +from ....kernels.ragged_ops import RaggedEmbeddingKernel +from ....ragged import RaggedBatchWrapper +from ...interfaces import DSEmbeddingBase, DSEmbeddingRegistry +from ...configs import DSEmbeddingsConfig + + +@DSEmbeddingRegistry.register_module +class DSRaggedEmbedding(DSEmbeddingBase): + + @staticmethod + def name(): + return 'ragged_embedding' + + @staticmethod + def supports_config(config: DSEmbeddingsConfig) -> bool: + + if DtypeEnum(config.residual_dtype) not in [DtypeEnum.fp16, DtypeEnum.bf16, DtypeEnum.fp32]: + return False + + if config.use_token_type: + return False + + if config.output_normalization is not None: + return False + + try: + _ = RaggedEmbeddingKernel(config.residual_dtype, torch.int32, config.embedding_dim) + except ValueError: + return False + + return True + + def __init__(self, config: DSEmbeddingsConfig, implementation_config: Dict[str, Any]) -> None: + super().__init__(config, implementation_config) + + self.embed_offset = self._config.positional_offset + + # TODO(cmikeh2): How do we want to avoid the int32 vs int64 issue? + self._ragged_embed = RaggedEmbeddingKernel(self._config.residual_dtype, torch.int32, + self._config.embedding_dim) + + self._output = torch.empty((self._config.max_tokens, self._config.embedding_dim), + dtype=self._config.residual_dtype, + device=get_accelerator().current_device()) + + @property + def output(self) -> torch.Tensor: + return self._output + + def forward(self, + ragged_batch: RaggedBatchWrapper, + word_embeddings: torch.Tensor, + position_embeddings: Optional[torch.Tensor] = None) -> torch.Tensor: + """ + Parameters: + ragged_batch (RaggedBatchWrapper): The input ids and associated ragged batch metadata. + word_embeddings (torch.Tensor): The word embedding table + """ + output = empty_from(self._output, (ragged_batch.tensor_toks, self._config.embedding_dim)) + self._ragged_embed(output, + ragged_batch, + word_embeddings, + position_embed_weight=position_embeddings, + position_embed_offset=self.embed_offset) + return output diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..0501af54c4e6dfb89ee0db49d56e46809ae7945c --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__init__.py @@ -0,0 +1,7 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .blas_fp_linear import BlasFPLinear +from .quantized_linear import QuantizedWf6Af16Linear, fp_quantize diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a13e175255065b2a76a8c847b8379f180f538044 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__pycache__/blas_fp_linear.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__pycache__/blas_fp_linear.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..4dce042896d747c7cdd6d10760fefd8ad026ece6 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__pycache__/blas_fp_linear.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__pycache__/quantized_linear.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__pycache__/quantized_linear.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..27d7cfd6e35a07868003e6f80ac9d339f2f3c87b Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/__pycache__/quantized_linear.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/blas_fp_linear.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/blas_fp_linear.py new file mode 100644 index 0000000000000000000000000000000000000000..c58dab0b826b48ed5ac4b58217936a2f9de4ae29 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/blas_fp_linear.py @@ -0,0 +1,103 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any, Dict, Optional + +import torch + +from deepspeed.accelerator import get_accelerator +from ....allocator import empty_from +from ....inference_utils import is_gated +from ....kernels.core_ops import ( + BlasLibLinear, + CUDABiasActivation, + CUDAGatedActivation, +) + +from ...interfaces import DSLinearBase, DSLinearRegistry +from ...configs import DSLinearConfig +from ....inference_parameter import InferenceParameter + + +@DSLinearRegistry.register_module +class BlasFPLinear(DSLinearBase): + """ + Linear DSModule based on BLAS library and standalone bias + activation kernel implementation. + """ + + @staticmethod + def name(): + return 'blas_fp_linear' + + @staticmethod + def supports_config(config: DSLinearConfig) -> bool: + if config.input_dtype != config.output_dtype: + return False + + if config.input_dtype != torch.float16 and config.input_dtype != torch.bfloat16: + return False + + if is_gated(config.activation): + try: + _ = CUDAGatedActivation(config.out_channels, config.output_dtype, config.activation) + except ValueError: + return False + else: + try: + _ = CUDABiasActivation(config.out_channels, config.output_dtype, config.activation) + except ValueError: + return False + + return True + + def __init__(self, config: DSLinearConfig, implementation_config: Dict[str, Any]) -> None: + super().__init__(config, implementation_config) + + self._linear_impl = BlasLibLinear(self._config.input_dtype) + + if is_gated(config.activation): + self._is_gated = True + self._act_fn = CUDAGatedActivation(config.out_channels, config.output_dtype, config.activation) + self._double_buffer = torch.empty((config.max_tokens, config.out_channels * 2), + dtype=config.output_dtype, + device=get_accelerator().current_device()) + else: + self._is_gated = False + self._act_fn = CUDABiasActivation(config.out_channels, config.output_dtype, config.activation) + + self._output = torch.empty((config.max_tokens, config.out_channels), + dtype=config.output_dtype, + device=get_accelerator().current_device()) + + def transform_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Converts param to same data type as input and output. + + Parameters: + param (torch.Tensor): Weight or bias tensor. + """ + param = param.to(self._config.output_dtype) + return InferenceParameter.initialize(param) + + def forward(self, hidden_states: torch.Tensor, w: torch.Tensor, b: Optional[torch.Tensor] = None) -> torch.Tensor: + + output = empty_from(self._output, (hidden_states.shape[0], self._config.out_channels)) + + if self._is_gated: + staging_output = empty_from(self._double_buffer, (hidden_states.shape[0], self._config.out_channels * 2)) + self._linear_impl(staging_output, hidden_states, w) + self._act_fn(output, staging_output, b) + else: + self._linear_impl(output, hidden_states, w) + self._act_fn(output, b) + + return output + + @property + def output(self) -> torch.Tensor: + """ + Return the padded, pre-allocated output Tensor. + """ + return self._output diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/quantized_linear.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/quantized_linear.py new file mode 100644 index 0000000000000000000000000000000000000000..933cf55b2391b0132f392e770d0b747f357f8ca0 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/linear/quantized_linear.py @@ -0,0 +1,205 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any, Dict, Optional + +import torch + +from deepspeed.accelerator import get_accelerator +from deepspeed.ops.op_builder import InferenceCoreBuilder +from ....allocator import empty_from +from ....inference_utils import is_gated +from ....kernels.core_ops import ( + CUDAWf6Af16Linear, + CUDABiasActivation, + CUDAGatedActivation, +) + +from ...interfaces import DSLinearBase, DSLinearRegistry +from ...configs import DSLinearConfig +from ....inference_parameter import InferenceParameter + + +def fp_quantize(input: torch.FloatTensor, + num_bits: int = 6, + exp_bits: int = 3, + min_value: torch.FloatTensor = None, + max_value: torch.FloatTensor = None, + group_size: int = -1): + """ + Args: + inputs (`torch.FloatTensor`) + The input which needs to be quantized + num_bits (int, >=4) + Number of bits to use for quantization + exp_bits: + fp exp_bits + min_value/max_vlue (torch.FloatTensor) + Used for static activation quantization + group_size (int) N + The quantization block size, each N numbers has its own scaling + factor and off-site. -1 means use the last dim as the group_size + Returns: + quantized_fake_fp6 + The quantized weights, in fp16 format and contains fp6 value. + scales + Quantization scales + """ + + try: + from qtorch.quant import float_quantize + except ImportError: + raise ImportError("Please install qtorch to use this function") + + assert (min_value is None and max_value is None) or (min_value is not None and max_value is not None) + + assert input.dtype == torch.float16 + + orig_device = input.device + input = input.to(torch.float32).to(get_accelerator().current_device()) + if num_bits == 6 and exp_bits == 3: # this is default + q_range = 28 + else: + raise NotImplementedError + + man_bits = num_bits - exp_bits - 1 + input_shape = input.shape + + if group_size == -1: + group_size = input_shape[-1] + else: + # Only support per-channel quantization + raise NotImplementedError + num_groups = input.numel() // group_size + input = input.reshape(num_groups, -1) + + if min_value is None: + max_input = torch.amax(torch.abs(input), dim=-1).view(num_groups, -1) + else: + max_input = torch.max(min_value.abs(), max_value) # .view(-1) + scales = max_input / q_range # q_range + 1 + scales[scales == 0] = 1 # avoid zero scales + scaled_input = input / scales + + quantized_fake_fp6 = float_quantize(scaled_input, exp_bits, man_bits, rounding="nearest") + + quantized_fake_fp6 = quantized_fake_fp6.reshape(input_shape).contiguous().to(torch.float16).to(orig_device) + scales = scales.to(torch.float16).to(orig_device) + # Now the dequantized value is quantized_fake_fp6 * scales + + return quantized_fake_fp6, scales + + +@DSLinearRegistry.register_module +class QuantizedWf6Af16Linear(DSLinearBase): + """ + Linear DSModule for FP6 weight-only quantization kernel, where weight is FP6 + and activation is FP16. + """ + + @staticmethod + def name(): + return 'quantized_wf6af16_linear' + + @staticmethod + def supports_config(config: DSLinearConfig) -> bool: + if config.input_dtype != config.output_dtype: + return False + + # As for fp6 data items, they are packed and stored in a set of fp16 + # tensors. E.g., 8 fp6 data items are stored in 3 fp16 tensor. + if config.input_dtype != torch.float16: + return False + + if is_gated(config.activation): + try: + _ = CUDAGatedActivation(config.out_channels, config.output_dtype, config.activation) + except ValueError: + return False + else: + try: + _ = CUDABiasActivation(config.out_channels, config.output_dtype, config.activation) + except ValueError: + return False + + return True + + def __init__(self, config: DSLinearConfig, implementation_config: Dict[str, Any]) -> None: + super().__init__(config, implementation_config) + + self._linear_impl = CUDAWf6Af16Linear() + + if is_gated(config.activation): + # In the FP6 kernel implementation, the MatMul is W * A, where W is + # the weight and A is activation. M is the output channel size. + self.out_channels = self._config.out_channels * 2 + self.in_channels = self._config.in_channels + self._is_gated = True + self._act_fn = CUDAGatedActivation(config.out_channels, config.output_dtype, config.activation) + self._double_buffer = torch.empty((config.max_tokens, config.out_channels * 2), + dtype=config.output_dtype, + device=get_accelerator().current_device()) + else: + self.out_channels = self._config.out_channels + self.in_channels = self._config.in_channels + self._is_gated = False + self._act_fn = CUDABiasActivation(config.out_channels, config.output_dtype, config.activation) + + self._output = torch.empty((config.max_tokens, config.out_channels), + dtype=config.output_dtype, + device=get_accelerator().current_device()) + + self.inf_module = InferenceCoreBuilder().load() + self.inf_module.create_handle() + self.preprocess_weight = self.inf_module.preprocess_weight + + self.quantizer = fp_quantize + + def transform_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Converts param to same data type as input and output. + + Parameters: + param (torch.Tensor): Weight or bias tensor. + """ + # It expects that the quantization scales are store in the attribute `scales`. + + if param.ndim == 1: # bias, do nothing + return InferenceParameter.initialize(param) + + quantized_fake_fp6, scales = self.quantizer(param, num_bits=6, exp_bits=3) + + # This is for debugging, will delete before release. + assert (quantized_fake_fp6.dtype == torch.float16) + assert quantized_fake_fp6.shape[0] == self.out_channels + assert scales.numel() == self.out_channels + + weights_2bit, weights_4bit = self.preprocess_weight(quantized_fake_fp6) + + return InferenceParameter.initialize(weights_2bit, weights_4bit=weights_4bit, scales=scales) + + def forward(self, hidden_states: torch.Tensor, w: torch.Tensor, b: Optional[torch.Tensor] = None) -> torch.Tensor: + weights_2bit = w + weights_4bit = w.weights_4bit + scales = w.scales + output = empty_from(self._output, (hidden_states.shape[0], self._config.out_channels)) + if self._is_gated: + staging_output = empty_from(self._double_buffer, (hidden_states.shape[0], self.out_channels)) + self._linear_impl(staging_output, hidden_states, weights_2bit, weights_4bit, scales, self.out_channels, + hidden_states.shape[0], self.in_channels) + self._act_fn(output, staging_output, b) + else: + self._linear_impl(output, hidden_states, weights_2bit, weights_4bit, scales, self.out_channels, + hidden_states.shape[0], self.in_channels) + self._act_fn(output, b) + + return output + + @property + def output(self) -> torch.Tensor: + """ + Return the padded, pre-allocated output Tensor. + """ + return self._output diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..053ad5da77460974cbc34361dcd887c5ea2851d8 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .cutlass_multi_gemm import DSMultiGemmMoE diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..cb3905f9f83ee7a3b0ef5c1b11b3617e2fbb3a77 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/__pycache__/cutlass_multi_gemm.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/__pycache__/cutlass_multi_gemm.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2ce4ebcc533dab63da46ce8b7ea8153c3f7e429b Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/__pycache__/cutlass_multi_gemm.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/cutlass_multi_gemm.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/cutlass_multi_gemm.py new file mode 100644 index 0000000000000000000000000000000000000000..a9b01d1233cd9af3aa9608ffedcdac13a031e411 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/moe/cutlass_multi_gemm.py @@ -0,0 +1,249 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any, Dict, Optional, Tuple + +import torch + +from deepspeed.accelerator import get_accelerator +from ....allocator import empty_from +from ....inference_utils import ActivationType, is_gated +from ....kernels.core_ops import BlasLibLinear, CUDAGatedActivation +from ....kernels.ragged_ops import ( + MoEGather, + MoEScatter, + RaggedTopKGating, +) +from ....ragged import RaggedBatchWrapper + +from ...interfaces import DSMoEBase, DSMoERegistry +from ...configs import DSMoEConfig +from ....kernels.cutlass_ops import MoEGEMM +from ....inference_parameter import InferenceParameter + + +@DSMoERegistry.register_module +class DSMultiGemmMoE(DSMoEBase): + """ + MoE implementation based on the CUTLASS multi-GEMM. + """ + + @staticmethod + def name(): + return 'cutlass_multi_gemm_moe' + + @staticmethod + def supports_config(config: DSMoEConfig) -> bool: + if config.input_dtype != config.output_dtype: + return False + + if config.input_dtype != torch.float16 and config.input_dtype != torch.bfloat16: + return False + + if config.top_k != 1 and config.top_k != 2 and config.top_k != 4 and config.top_k != 8: + return False + + return True + + def __init__(self, config: DSMoEConfig, implementation_config: Dict[str, Any]) -> None: + super().__init__(config, implementation_config) + + # Convenience variables for frequently accessed items. + self.max_tokens = self._config.max_tokens + self.n_experts = self._config.n_experts + self.n_top_k = self._config.top_k + self.intermediate_dim = self._config.intermediate_features + + moe_op_act_fn = ActivationType.IDENTITY if is_gated(self._config.activation) else self._config.activation + + self._mlp_1 = MoEGEMM(fp_dtype=implementation_config['weight_dtype'], act_fn=moe_op_act_fn) + self._mlp_2 = MoEGEMM(fp_dtype=implementation_config['weight_dtype'], act_fn=ActivationType.IDENTITY) + + if is_gated(self._config.activation): + self._activation = CUDAGatedActivation(self._config.model_dim, self._config.input_dtype, + self._config.activation) + else: + self._activation = None + + self._gate_proj = BlasLibLinear(self._config.input_dtype) + self._top_1_gate = RaggedTopKGating(config.input_dtype) + self._moe_scatter = MoEScatter(config.input_dtype, config.model_dim) + self._moe_gather = MoEGather(config.input_dtype, config.model_dim, config.normalize_scores) + + self._create_buffers() + + def _create_buffers(self): + + # Gating buffers + self._logits = torch.empty((self._config.max_tokens, self.n_experts), + dtype=self._config.input_dtype, + device=get_accelerator().current_device()) + self._expert_counts = torch.empty((self.n_experts, ), + dtype=torch.int32, + device=get_accelerator().current_device()) + self._scores = torch.empty((self._config.max_tokens, self.n_top_k), + dtype=torch.float32, + device=get_accelerator().current_device()) + self._assignments = torch.empty((self._config.max_tokens, self.n_top_k), + dtype=torch.int32, + device=get_accelerator().current_device()) + self._offsets = torch.empty((self._config.max_tokens, self.n_top_k), + dtype=torch.int32, + device=get_accelerator().current_device()) + + # Scatter buffers + self._moe_input = torch.empty((self._config.max_tokens * self.n_top_k, self._config.model_dim), + dtype=self._config.input_dtype, + device=get_accelerator().current_device()) + self._expert_cumsum = torch.empty((self._config.n_experts, ), + dtype=torch.int64, + device=get_accelerator().current_device()) + self._mapped_slots = torch.empty((self._config.max_tokens, self.n_top_k), + dtype=torch.int32, + device=get_accelerator().current_device()) + + # GEMM Buffers + self._intermediate = torch.empty((self._config.max_tokens * self.n_top_k, self._config.intermediate_features), + dtype=self._config.output_dtype, + device=get_accelerator().current_device()) + if self._activation is not None: + self._gated_intermediate = torch.empty( + (self._config.max_tokens * self.n_top_k, self._config.intermediate_features * 2), + dtype=self._config.output_dtype, + device=get_accelerator().current_device()) + + self._output_unordered = torch.empty((self._config.max_tokens * self.n_top_k, self._config.model_dim), + dtype=self._config.output_dtype, + device=get_accelerator().current_device()) + + # Gather buffer + self._output = torch.empty((self._config.max_tokens, self._config.model_dim), + dtype=self._config.output_dtype, + device=get_accelerator().current_device()) + + def transform_gate_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Ensures gate param is going to match the activation data type. + """ + param = param.to(self._config.input_dtype) + return InferenceParameter.initialize(param) + + def transform_moe_mlp_1_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Converts param to same data type as input and output. + + Parameters: + param (torch.Tensor): Weight or bias tensor. + """ + param = param.to(self._config.input_dtype) + + if len(param.shape) == 3: + param = param.permute(0, 2, 1).contiguous() + return InferenceParameter.initialize(param) + + def transform_moe_mlp_2_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Converts param to same data type as input and output. + + Parameters: + param (torch.Tensor): Weight or bias tensor. + """ + param = param.to(self._config.input_dtype) + + if len(param.shape) == 3: + param = param.permute(0, 2, 1).contiguous() + return InferenceParameter.initialize(param) + + @property + def output(self) -> torch.Tensor: + return self._output + + def _gate(self, hidden_states: torch.Tensor, batch_metadata: RaggedBatchWrapper, + gate_w: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]: + """ + Helper function to isolate the logit for gating. This will take the hidden states + and produce the metadata + tensors for the CUTLASS ragged GEMMs. If the input has + been padded for CG, this will strip the padding for MoE. + + Parameters: + hidden_states (torch.Tensor): Hidden states tensor. Expected shape is [n_tokens, model_dim]. + batch_metadata (RaggedBatchWrapper): Batch metadata for the hidden states. + gate_w (torch.Tensor): Gate weight tensor. Expected shape is [num_experts, model_dim]. + + Returns: + Tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]: The MoE input, the cumsum of the offsets (for the MoE kernels themselves), the scores, and the mapped slots (to recover the original order of the tokens) + """ + + # Get views on the buffers for gating + logits = empty_from(self._logits, (hidden_states.shape[0], self._logits.shape[-1])) + scores = empty_from(self._scores, (hidden_states.shape[0], self.n_top_k)) + assignments = empty_from(self._assignments, (hidden_states.shape[0], self.n_top_k)) + offsets = empty_from(self._offsets, (hidden_states.shape[0], self.n_top_k)) + mapped_slots = empty_from(self._mapped_slots, (hidden_states.shape[0], self.n_top_k)) + moe_input = empty_from(self._moe_input, (hidden_states.shape[0] * self.n_top_k, self._moe_input.shape[-1])) + + self._gate_proj(logits, hidden_states, gate_w) + self._expert_counts.zero_() + self._top_1_gate(self._expert_counts, scores, assignments, offsets, logits, batch_metadata) + self._moe_scatter(moe_input, self._expert_cumsum, mapped_slots, hidden_states, self._expert_counts, + assignments, offsets) + + return moe_input, self._expert_cumsum, scores, mapped_slots + + def forward(self, + hidden_states: torch.Tensor, + batch_metadata: RaggedBatchWrapper, + gate_w: torch.Tensor, + mlp_1_w: torch.Tensor, + mlp_2_w: torch.Tensor, + mlp_1_b: Optional[torch.Tensor] = None, + mlp_2_b: Optional[torch.Tensor] = None) -> torch.Tensor: + """ + MoE forward pass built on top of CUTLASS multi-GEMM. + + Parameters: + hidden_states (torch.Tensor): Hidden states tensor. Expected shape is [batch, seq_len, model_dim]. + gate_w (torch.Tensor): Gate weight tensor. Expected shape is [num_experts, model_dim]. + """ + + moe_input, expert_cumsum, scores, mapped_slots = self._gate(hidden_states, batch_metadata, gate_w) + + # Get views on the buffers for GEMM + intermediate = empty_from(self._intermediate, + (hidden_states.shape[0] * self.n_top_k, self._intermediate.shape[-1])) + output_unordered = empty_from(self._output_unordered, + (hidden_states.shape[0] * self.n_top_k, self._output_unordered.shape[-1])) + output = empty_from(self._output, (hidden_states.shape[0], self._output.shape[-1])) + + if self._activation is not None: + gated_intermediate = empty_from( + self._gated_intermediate, (hidden_states.shape[0] * self.n_top_k, self._gated_intermediate.shape[-1])) + self._mlp_1( + gated_intermediate, + moe_input, + mlp_1_w, + expert_cumsum, + mlp_1_b, + ) + self._activation(intermediate, gated_intermediate) + else: + self._mlp_1( + intermediate, + moe_input, + mlp_1_w, + expert_cumsum, + mlp_1_b, + ) + + self._mlp_2( + output_unordered, + intermediate, + mlp_2_w, + expert_cumsum, + mlp_2_b, + ) + + self._moe_gather(output, output_unordered, scores, mapped_slots, self._expert_counts) + return output diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..653a2fe4fb5be742c873aa51e76db79ead7325df --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .cuda_post_ln import DSPostLNCUDAModule diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..74d18879129619d9af4fff1ea0dec181742cb752 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/__pycache__/cuda_post_ln.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/__pycache__/cuda_post_ln.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..8dc345c71ef90d903b1e9494e6222e50d1eec74d Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/__pycache__/cuda_post_ln.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/cuda_post_ln.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/cuda_post_ln.py new file mode 100644 index 0000000000000000000000000000000000000000..9b2af4bb90231cd3f37e3406fe94042c4006faa3 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/post_norm/cuda_post_ln.py @@ -0,0 +1,56 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any, Dict, Tuple + +import torch + +from deepspeed.accelerator import get_accelerator +from ...interfaces import DSPostNormBase, DSPostNormRegistry +from ...configs import DSNormConfig +from ....kernels.core_ops.cuda_layer_norm.cuda_post_ln import CUDAFPPostLN +from ....allocator import empty_from +from ....inference_parameter import InferenceParameter + + +@DSPostNormRegistry.register_module +class DSPostLNCUDAModule(DSPostNormBase): + + @staticmethod + def name(): + return 'cuda_post_ln' + + @staticmethod + def supports_config(config: DSNormConfig): + if len(set([config.residual_dtype, config.input_dtype, config.output_dtype])) != 1: + return False + + try: + _ = CUDAFPPostLN(config.channels, config.residual_dtype) + except ValueError: + return False + return True + + def __init__(self, config: DSNormConfig, implementation_config: Dict[str, Any]): + super().__init__(config, implementation_config) + self._fp_post_ln = CUDAFPPostLN(self._config.channels, self._config.residual_dtype, epsilon=self._config.eps) + + self._output = torch.empty((config.max_tokens, config.channels), + dtype=config.output_dtype, + device=get_accelerator().current_device()) + + def transform_param(self, param: torch.Tensor) -> InferenceParameter: + param = param.to(self._config.input_dtype) + return InferenceParameter.initialize(param) + + def forward(self, residual: torch.Tensor, hidden_in: torch.Tensor, gamma: torch.Tensor, + beta: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Since the CUDA FP only supports all data types being the same, we will alias the residual + with our output. + """ + self._residual_output = empty_from(self._output, residual.shape) + self._fp_post_ln(residual, residual, hidden_in, gamma, beta) + return residual, residual diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..12605f13f955e00b598516a97a8cc7d0369f7405 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__init__.py @@ -0,0 +1,7 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .cuda_pre_ln import DSPreLNCUDAModule +from .cuda_pre_rms import DSPreRMSCUDAModule diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..0b75c06d33d517d6871e95cf0edf4aa476f0bf7e Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__pycache__/cuda_pre_ln.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__pycache__/cuda_pre_ln.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..aaf5e94b5ea9416d4d31053674d3d4ea8a5add36 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__pycache__/cuda_pre_ln.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__pycache__/cuda_pre_rms.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__pycache__/cuda_pre_rms.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..163a1f8c7598824cf24f09ae4082afbb7b10e465 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/__pycache__/cuda_pre_rms.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/cuda_pre_ln.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/cuda_pre_ln.py new file mode 100644 index 0000000000000000000000000000000000000000..90783ce8c9a63320fc2047b4e8dfed9522df7899 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/cuda_pre_ln.py @@ -0,0 +1,69 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any, Dict, Optional, Tuple + +import torch + +from deepspeed.accelerator import get_accelerator +from ...interfaces import DSPreNormBase, DSPreNormRegistry +from ...configs import DSNormConfig, NormTypeEnum +from ....kernels.core_ops.cuda_layer_norm.cuda_pre_ln import CUDAFPPreLN +from ....kernels.core_ops.cuda_layer_norm.cuda_ln import CUDAFPLN +from ....allocator import empty_from +from ....inference_parameter import InferenceParameter + + +@DSPreNormRegistry.register_module +class DSPreLNCUDAModule(DSPreNormBase): + + @staticmethod + def name(): + return 'cuda_pre_ln' + + @staticmethod + def supports_config(config: DSNormConfig): + type = NormTypeEnum(config.type) + if type != NormTypeEnum.LayerNorm: + return False + + if len(set([config.residual_dtype, config.input_dtype, config.output_dtype])) != 1: + return False + + try: + _ = CUDAFPPreLN(config.channels, config.residual_dtype) + except ValueError: + return False + return True + + def __init__(self, config: DSNormConfig, implementation_config: Dict[str, Any]): + super().__init__(config, implementation_config) + self._fp_pre_ln = CUDAFPPreLN(self._config.channels, self._config.residual_dtype, epsilon=self._config.eps) + self._fp_ln = CUDAFPLN(self._config.channels, self._config.residual_dtype, epsilon=self._config.eps) + + # Buffers for the hidden output (residual is updated in-place) + self._hidden_output = torch.empty((config.max_tokens, config.channels), + dtype=config.output_dtype, + device=get_accelerator().current_device()) + + def transform_param(self, param: torch.Tensor) -> InferenceParameter: + param = param.to(self._config.input_dtype) + return InferenceParameter.initialize(param) + + def forward(self, residual: torch.Tensor, hidden_in: Optional[torch.Tensor], gamma: torch.Tensor, + beta: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Since the CUDA FP only supports all data types being the same, we will alias the residual + with our output. + + If hidden_in is None, that means we do not need to perform the residual add and will + only return the hidden output modified. + """ + hidden_out = empty_from(self._hidden_output, residual.shape) + if hidden_in is None: + self._fp_ln(hidden_out, residual, gamma, beta) + else: + self._fp_pre_ln(residual, hidden_out, residual, hidden_in, gamma, beta) + return residual, hidden_out diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/cuda_pre_rms.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/cuda_pre_rms.py new file mode 100644 index 0000000000000000000000000000000000000000..986262b31b1f50c3c139e7f71aca94c13ed866c3 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/pre_norm/cuda_pre_rms.py @@ -0,0 +1,79 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any, Dict, Optional, Tuple + +import torch + +from deepspeed.accelerator import get_accelerator +from ...interfaces import DSPreNormBase, DSPreNormRegistry +from ...configs import DSNormConfig, NormTypeEnum +from ....kernels.core_ops import CUDARMSNorm, CUDARMSPreNorm +from ....allocator import empty_from +from ....inference_parameter import InferenceParameter + + +@DSPreNormRegistry.register_module +class DSPreRMSCUDAModule(DSPreNormBase): + + @staticmethod + def name(): + return 'cuda_pre_rms' + + @staticmethod + def supports_config(config: DSNormConfig): + type = NormTypeEnum(config.type) + if type != NormTypeEnum.RMSNorm: + return False + + if len(set([config.residual_dtype, config.input_dtype, config.output_dtype])) != 1: + return False + + try: + # Only need to check one since the support matrix for the two rms kernels is the same + _ = CUDARMSPreNorm(config.channels, config.residual_dtype) + except ValueError: + return False + return True + + def __init__(self, config: DSNormConfig, implementation_config: Dict[str, Any]): + super().__init__(config, implementation_config) + self._fp_rms = CUDARMSNorm(self._config.channels, self._config.residual_dtype, epsilon=self._config.eps) + self._fp_rms_pre = CUDARMSPreNorm(self._config.channels, self._config.residual_dtype, epsilon=self._config.eps) + + # Buffers for both the hidden and residual outputs + self._hidden_output = torch.empty((config.max_tokens, config.channels), + dtype=config.output_dtype, + device=get_accelerator().current_device()) + self._residual_output = torch.empty((config.max_tokens, config.channels), + dtype=config.output_dtype, + device=get_accelerator().current_device()) + + def transform_param(self, param: torch.Tensor) -> InferenceParameter: + param = param.to(self._config.input_dtype) + return InferenceParameter.initialize(param) + + def forward(self, + residual: torch.Tensor, + hidden_in: Optional[torch.Tensor], + gamma: torch.Tensor, + beta: Optional[torch.Tensor] = None) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Since the CUDA FP only supports all data types being the same, we will alias the residual + with our output. + + If hidden_in is None, that means we do not need to perform the residual add and will + only return the hidden output modified. + """ + assert beta is None, "Beta is not supported for RMSNorm" + + hidden_out = empty_from(self._hidden_output, residual.shape) + if hidden_in is None: + self._fp_rms(hidden_out, residual, gamma) + residual_out = residual + else: + residual_out = empty_from(self._residual_output, residual.shape) + self._fp_rms_pre(residual_out, hidden_out, residual, hidden_in, gamma) + return residual_out, hidden_out diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..4a5fd24d518bbf29f88771478b04da73399d4a53 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/__init__.py @@ -0,0 +1,6 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .ragged_unembed import DSRaggedUnembed diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d509919a997d33b866c0b20c1eb147e255af82e3 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/__pycache__/ragged_unembed.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/__pycache__/ragged_unembed.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..303286660e2f8c5cc5aa07c3d0bddc4c59d51606 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/__pycache__/ragged_unembed.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/ragged_unembed.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/ragged_unembed.py new file mode 100644 index 0000000000000000000000000000000000000000..36130902c665c7adff419ef0ea2be7138292e193 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/implementations/unembed/ragged_unembed.py @@ -0,0 +1,120 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any, Dict, Optional + +import torch + +from deepspeed.accelerator import get_accelerator +from ....allocator import empty_from +from ....inference_utils import DtypeEnum, ActivationType +from ....kernels.core_ops import CUDAFPLN, BlasLibLinear, CUDARMSNorm, CUDABiasActivation +from ....kernels.ragged_ops import RaggedLogitsGather +from ....ragged import RaggedBatchWrapper +from ...interfaces import DSUnembedBase, DSUnembedRegistry +from ...configs import DSUnembedConfig + + +@DSUnembedRegistry.register_module +class DSRaggedUnembed(DSUnembedBase): + """ + Ragged unembedding implementation. This implementation will gather only the last token + of each sequence in the ragged inflight batch and calculate the logits only for those rows. + """ + + @staticmethod + def name(): + return 'ragged_unembed' + + @staticmethod + def supports_config(config: DSUnembedConfig): + + if DtypeEnum(config.dtype) not in [DtypeEnum.fp16, DtypeEnum.bf16, DtypeEnum.fp32]: + return False + + try: + _ = RaggedLogitsGather(config.model_dim, config.dtype) + except ValueError: + return False + + if config.norm_type == 'rms_norm': + try: + _ = CUDARMSNorm(config.model_dim, config.dtype) + except ValueError: + return False + elif config.norm_type == 'layer_norm': + try: + _ = CUDAFPLN(config.model_dim, config.dtype) + except ValueError: + return False + + return True + + def __init__(self, config: DSUnembedConfig, implementation_config: Dict[str, Any]) -> None: + super().__init__(config, implementation_config) + + self._logits_gather = RaggedLogitsGather(config.model_dim, self._config.dtype) + + if self._config.norm_type == 'layer_norm': + self._norm = CUDAFPLN(self._config.model_dim, self._config.dtype) + elif self._config.norm_type == 'rms_norm': + self._norm = CUDARMSNorm(self._config.model_dim, self._config.dtype) + else: + self._norm = None + + self._linear = BlasLibLinear(self._config.dtype) + # Here the activation kernel is being used to apply bias, hence the identity activation type! + self._act_fn = CUDABiasActivation(self._config.vocab_size, self._config.dtype, ActivationType.IDENTITY) + + self._intermediate = torch.empty((self._config.max_sequences, self._config.model_dim), + dtype=self._config.dtype, + device=get_accelerator().current_device()) + + self._output = torch.empty((self._config.max_sequences, self._config.vocab_size), + dtype=self._config.dtype, + device=get_accelerator().current_device()) + + @property + def output(self) -> torch.Tensor: + return self._output + + def forward(self, + hidden_states: torch.Tensor, + vocab_embedding: torch.Tensor, + ragged_metadata: RaggedBatchWrapper, + bias: Optional[torch.Tensor] = None, + gamma: Optional[torch.Tensor] = None, + beta: Optional[torch.Tensor] = None) -> torch.Tensor: + """ + Return final model logits. + + Args: + hidden_states (torch.Tensor): The hidden states from the model. This is the output of the + final layer of the model. + vocab_embedding (torch.Tensor): The vocab embedding table. + raged_metadata (RaggedBatchWrapper): The ragged batch metadata. + gamma (Optional[torch.Tensor]): The gamma tensor for normalization. + beta (Optional[torch.Tensor]): The beta tensor for normalization. + """ + + cut_down_hidden_states = empty_from(self._intermediate, + (ragged_metadata.current_sequences, self._config.model_dim)) + self._logits_gather(cut_down_hidden_states, hidden_states, ragged_metadata) + + if self._config.norm_type == 'rms_norm': + if gamma is None: + raise ValueError('RMS Normalization enabled but gamma not provided.') + self._norm(cut_down_hidden_states, cut_down_hidden_states, gamma) + elif self._config.norm_type == 'layer_norm': + if gamma is None or beta is None: + raise ValueError('Normalization enabled but gamma and/or beta not provided.') + self._norm(cut_down_hidden_states, cut_down_hidden_states, gamma, beta) + + output = empty_from(self._output, (ragged_metadata.current_sequences, self._config.vocab_size)) + self._linear(output, cut_down_hidden_states, vocab_embedding) + if bias is not None: + self._act_fn(output, bias) + + return output diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..13b556789e4e43c4e816319cd48e95e9e51afb0c --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__init__.py @@ -0,0 +1,12 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .attention_base import DSSelfAttentionRegistry, DSSelfAttentionBase +from .embedding_base import DSEmbeddingRegistry, DSEmbeddingBase +from .linear_base import DSLinearRegistry, DSLinearBase +from .moe_base import DSMoERegistry, DSMoEBase +from .post_norm_base import DSPostNormRegistry, DSPostNormBase +from .pre_norm_base import DSPreNormRegistry, DSPreNormBase +from .unembed_base import DSUnembedRegistry, DSUnembedBase diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..fd3b442eb2d6a2a6644430acc8eb95ffd989f4ae Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/attention_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/attention_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..90bdfb9ed38d9a7935583f801894da55580a3beb Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/attention_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/embedding_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/embedding_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..8d72bdba4c0c8c22edbe65e6ee79dd64954b1f9d Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/embedding_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/linear_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/linear_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..bca728ad046e7ab4c8f8a8ce1d3750f1ec4c44ff Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/linear_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/moe_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/moe_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d0ee7ece63f72ef37ec17d265cd558c74e0dcf7c Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/moe_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/post_norm_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/post_norm_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e569290a91a47648fa1f8e1c3c75fab9f0fd2a79 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/post_norm_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/pre_norm_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/pre_norm_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..45748b40dfa131d8ecf67a3a8c6780558e2d34c3 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/pre_norm_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/unembed_base.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/unembed_base.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f589d510fcc8fb54bebe61b0187838ed42756b64 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/__pycache__/unembed_base.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/attention_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/attention_base.py new file mode 100644 index 0000000000000000000000000000000000000000..c67dc033f92ada844c1c5d27bde5ac0cf7f3f0ac --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/attention_base.py @@ -0,0 +1,97 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any, Dict, Optional, Tuple, Type + +import torch + +from ...ragged import RaggedBatchWrapper +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from ..ds_module import DSModuleBase +from ..module_registry import DSModuleRegistryBase +from ..configs import DSSelfAttentionConfig + + +class DSSelfAttentionBase(DSModuleBase): + """ + Base mixin for all attention modules. The interface represented by this module + is broadly: + + output = attention(query_key_value, + Optional[kv_cache], + Optional[attention_mask], + Optional[attention_bias]) + """ + + @staticmethod + def config_class() -> Type[DeepSpeedConfigModel]: + return DSSelfAttentionConfig + + def __init__(self, config: DSSelfAttentionConfig, implementation_config: Dict[str, Any]) -> None: + super().__init__(config, implementation_config) + + @property + def kv_block_size(self) -> int: + """ + Return preferred granulatity for blocked KV-cache implementation. + """ + raise NotImplementedError() + + @property + def q_block_size(self) -> int: + """ + Property to calculate blocking granularity for the query dimension. + This has no impact on the KV-cache structure, but will affect the + number of attention atoms associated with a batch. + """ + raise NotImplementedError() + + def build_atoms(self, ragged_batch: RaggedBatchWrapper) -> None: + """ + Build the atoms for this module. This is not a strict requirement for the class, + so this method is a no-op by default rather than abstract. + """ + pass + + def forward(self, + q_k_v: torch.Tensor, + kv_cache: torch.Tensor, + batch: RaggedBatchWrapper, + attention_mask: Optional[torch.Tensor] = None, + attention_bias: Optional[torch.Tensor] = None, + inv_freqs: Optional[torch.Tensor] = None) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Parameters: + q_k_v (torch.Tensor): Query, key, and value tensors. Expected shape is: + [ + batch, + seq_len, + 2 * self._config.n_heads_kv + self._config.n_heads_q, + self._config.head_size + ]. + kv_cache (Optional[torch.Tensor]): Key and value cache tensor. Expected shape is + [ + 2, + batch, + kv_cache_len, + self._config.n_heads_kv, + self._config.head_size + ]. If None, cache is disabled. The `kv_cache_len` dimension does not need to + be contiguous (it should expand stride by `max_out_tokens`). + batch (RaggedBatchWrapper): Ragged batch metadata. + attention_mask (Optional[torch.Tensor]): Attention mask tensor. If None, masking is + disabled. This will defer to the config in the case of conflicting information. + This means if the config class is implying causal attention, the mask will be ignored. + attention_bias (Optional[torch.Tensor]): Attention bias tensor. If None, bias is disabled. + """ + raise NotImplementedError() + + +class DSSelfAttentionRegistry(DSModuleRegistryBase): + registry: Dict = {} + + @staticmethod + def associated_class() -> Type[DSModuleBase]: + return DSSelfAttentionBase diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/embedding_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/embedding_base.py new file mode 100644 index 0000000000000000000000000000000000000000..1ab7e5f0b7a2455ad3e60bb12a9d262900340e45 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/embedding_base.py @@ -0,0 +1,85 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from abc import abstractmethod +from typing import Any, Dict, Optional, Type + +import torch + +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from ...ragged import RaggedBatchWrapper +from ..ds_module import DSModuleBase +from ..module_registry import DSModuleRegistryBase +from ..configs import DSEmbeddingsConfig +from ...inference_parameter import InferenceParameter + + +class DSEmbeddingBase(DSModuleBase): + """ + Base mixin for embedding modules. The interface represented by this module is: + + hidden_out = embedding(input_ids) + + position_embedding(position_ids) + + token_type_embedding(token_type_ids) + with optional normalization. + """ + + @staticmethod + def config_class() -> Type[DeepSpeedConfigModel]: + return DSEmbeddingsConfig + + def __init__(self, config: DSEmbeddingsConfig, implementation_config: Dict[str, Any]) -> None: + super().__init__(config, implementation_config) + + def transform_param(self, embed_param: torch.Tensor) -> InferenceParameter: + """ + Perform any necessary transformations on an embedding parameter. This module assumes + that all embedding parameters would require the same set of transformations. + + Parameters: + embed_param (torch.Tensor): Embedding parameter. Shape is of [vocab_size, hidden_size] + """ + raise NotImplementedError() + + @property + @abstractmethod + def output(self) -> torch.Tensor: + """ + Pre-allocated output Tensor. This currently needs to be exposed for gather operations + on the output. + + TODO(cmikeh2): This is not ideal. We need a better abstraction for this, such as giving + access to the inference comm object to the DSModule. + """ + raise NotImplementedError() + + def forward(self, + ragged_batch: RaggedBatchWrapper, + word_embeddings: torch.Tensor, + position_embeddings: Optional[torch.Tensor] = None, + token_type_ids: Optional[torch.Tensor] = None, + token_type_embeddings: Optional[torch.Tensor] = None) -> InferenceParameter: + """ + Parameters: + ragged_batch (torch.Tensor): Ragged batch of token ids + associated metadata. + word_embeddings (torch.Tensor): Word embeddings. + position_embeddings (torch.Tensor): Position embeddings. If passed, IDs will be + inferred from the ragged batch itself. + token_type_ids (torch.Tensor): Token type ids. + token_type_embeddings (torch.Tensor): Token type embeddings. + + Returns: + torch.Tensor: Hidden states. This should be the sum of the relevant + encodings for the model. + """ + raise NotImplementedError() + + +class DSEmbeddingRegistry(DSModuleRegistryBase): + registry: Dict = {} + + @staticmethod + def associated_class() -> Type[DSModuleBase]: + return DSEmbeddingBase diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/linear_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/linear_base.py new file mode 100644 index 0000000000000000000000000000000000000000..fe6ccbcd934490c94a0b656f09b1f25aef70c163 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/linear_base.py @@ -0,0 +1,72 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from abc import abstractmethod +from typing import Any, Dict, Optional, Type + +import torch + +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from ..ds_module import DSModuleBase +from ..module_registry import DSModuleRegistryBase +from ..configs import DSLinearConfig +from ...inference_parameter import InferenceParameter + + +class DSLinearBase(DSModuleBase): + """ + Base mixin for all Linear modules. The interface represented by this module + is: + + hidden_out = activation(hidden_in * weight + bias) + + The format and dtype of the weight and bias tensors are not defined and implementations + may compress as necessary. Must support a bias. + """ + + @staticmethod + def config_class() -> Type[DeepSpeedConfigModel]: + return DSLinearConfig + + def __init__(self, config: DSLinearConfig, implementation_config: Dict[str, Any]) -> None: + super().__init__(config, implementation_config) + + @abstractmethod + def transform_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Perform any necessary transformations of the parameters of this module. + + Parameters: + param (torch.Tensor): Weight or bias tensor. + """ + ... + + def forward(self, hidden_states: torch.Tensor, w: torch.Tensor, b: Optional[torch.Tensor] = None) -> torch.Tensor: + """ + Parameters: + hidden_states (torch.Tensor): Hidden states tensor. Expected shape is either + [batch, seq_len, in_channels] or [batch, in_channels]. + + Returns: + torch.Tensor: Output tensor. Tensor should have same number of dimensions as + input tensor. + """ + raise NotImplementedError() + + @property + @abstractmethod + def output(self) -> torch.Tensor: + """ + Return the padded, pre-allocated output Tensor. + """ + ... + + +class DSLinearRegistry(DSModuleRegistryBase): + registry: Dict = {} + + @staticmethod + def associated_class() -> Type[DSModuleBase]: + return DSLinearBase diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/moe_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/moe_base.py new file mode 100644 index 0000000000000000000000000000000000000000..78bdc0700f63b9aa532fc62b66b3e9c19261061c --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/moe_base.py @@ -0,0 +1,91 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from abc import abstractmethod +from typing import Any, Dict, Optional, Type + +import torch + +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from ..ds_module import DSModuleBase +from ..module_registry import DSModuleRegistryBase +from ..configs import DSMoEConfig +from ...inference_parameter import InferenceParameter + + +class DSMoEBase(DSModuleBase): + """ + Base mixing for MoE modules. The interface represented by this module is: + + expert_assignments = gate(hidden_states) + intermediate = ragged_linear(hidden_states, expert_assignments) + output = ragged_linear(intermediate, expert_assignments) + """ + + @staticmethod + def config_class() -> Type[DeepSpeedConfigModel]: + return DSMoEConfig + + def __init__(self, config: DSMoEConfig, implementation_config: Dict[str, Any]) -> None: + super().__init__(config, implementation_config) + + @abstractmethod + def transform_gate_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Perform any necessary transformations of the gate parameter. + + Args: + param (torch.Tensor): gate_w (shape: [num_experts, model_dim]) + """ + ... + + @abstractmethod + def transform_moe_mlp_1_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Perform any necessary transformations of the parameter. The specific component + being transformed should be inferred from the shape of the parameter. + + Args: + param (torch.Tensor): One of either mlp_1_w, mlp_1_b + """ + ... + + @abstractmethod + def transform_moe_mlp_2_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Perform any necessary transformations of the parameter. The specified component being + transformed should be inferred from the shape of the parameter. This interface is + separate from transform_moe_1_param because the two components may have identical + shapes. + + Args: + param (torch.Tensor): One of either mlp_2_w or mlp_2_b + """ + ... + + def forward(self, + hidden_states: torch.Tensor, + gate_w: torch.Tensor, + mlp_1_w: torch.Tensor, + mlp_2_w: torch.Tensor, + mlp_1_b: Optional[torch.Tensor] = None, + mlp_2_b: Optional[torch.Tensor] = None) -> torch.Tensor: + raise NotImplementedError() + + @property + @abstractmethod + def output(self) -> torch.Tensor: + """ + Returns the pre-allocated, padded output Tensor. + """ + ... + + +class DSMoERegistry(DSModuleRegistryBase): + registry: Dict = {} + + @staticmethod + def associated_class() -> Type[DSModuleBase]: + return DSMoEBase diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/post_norm_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/post_norm_base.py new file mode 100644 index 0000000000000000000000000000000000000000..cc80e5c94bf73e7e2173906bd220fc39ff7df610 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/post_norm_base.py @@ -0,0 +1,69 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from abc import abstractmethod +from typing import Any, Dict, Optional, Tuple, Type + +import torch + +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from ..ds_module import DSModuleBase +from ..configs.norm_config import DSNormConfig +from ..module_registry import DSModuleRegistryBase +from ...inference_parameter import InferenceParameter + + +class DSPostNormBase(DSModuleBase): + """ + Base MixIn for all Post-Normalization modules. The interface represented by this + module is: + + residual, hidden_out = norm(residual + hidden_in) + + If residual and hidden_out are the same data type, then they may alias each other. + Furthermore, residual should be updated in-place. + """ + + @staticmethod + def config_class() -> Type[DeepSpeedConfigModel]: + return DSNormConfig + + def __init__(self, config: DSNormConfig, implementation_config: Dict[str, Any]) -> None: + super().__init__(config, implementation_config) + + @abstractmethod + def transform_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Transform a gamma/beta parameter. It is assumed that both transformations are + the same. + + Parameters: + param (torch.Tensor): Gamma or beta parameter. + """ + ... + + def forward(self, + residual: torch.Tensor, + hidden_states: torch.Tensor, + gamma: torch.Tensor, + beta: Optional[torch.Tensor] = None) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Parameters: + residual (torch.Tensor): Residual tensor. + hidden_states (torch.Tensor): Hidden states tensor. + + Returns: + (torch.Tensor, torch.Tensor): Tuple of residual and hidden states. + Hidden states may alias with residual. + """ + raise NotImplementedError() + + +class DSPostNormRegistry(DSModuleRegistryBase): + registry: Dict = {} + + @staticmethod + def associated_class() -> Type[DSModuleBase]: + return DSPostNormBase diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/pre_norm_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/pre_norm_base.py new file mode 100644 index 0000000000000000000000000000000000000000..84f51cff6947ecf8ee8563c289aec39324f37e7b --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/pre_norm_base.py @@ -0,0 +1,73 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from abc import abstractmethod +from typing import Any, Dict, Optional, Tuple, Type + +import torch + +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from ..ds_module import DSModuleBase +from ..configs.norm_config import DSNormConfig +from ..module_registry import DSModuleRegistryBase +from ...inference_parameter import InferenceParameter + + +class DSPreNormBase(DSModuleBase): + """ + Base mixin for all Pre-Normalization modules. The interface represented by this module + is: + + if hidden_in is not None: + residual_out = residual + hidden_in + else: + residual_out = residual + + hidden_out = normalize(residual_out) + return residual_out, hidden_out + + Residual should be updated in-place. + """ + + @staticmethod + def config_class() -> Type[DeepSpeedConfigModel]: + return DSNormConfig + + def __init__(self, config: DSNormConfig, implementation_config: Dict[str, Any]): + super().__init__(config, implementation_config) + + @abstractmethod + def transform_param(self, param: torch.Tensor) -> InferenceParameter: + """ + Transform a gamma/beta parameter. It is assumed that both transformations are + the same. + + Parameters: + param (torch.Tensor): Gamma or beta parameter. + """ + ... + + def forward(self, + residual: torch.Tensor, + hidden_states: Optional[torch.Tensor], + gamma: torch.Tensor, + beta: Optional[torch.Tensor] = None) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Parameters: + residual (torch.Tensor): Residual tensor. + hidden_states (torch.Tensor): Hidden states tensor. + + Returns: + (torch.Tensor, torch.Tensor): Tuple of residual and hidden states. + """ + raise NotImplementedError() + + +class DSPreNormRegistry(DSModuleRegistryBase): + registry: Dict = {} + + @staticmethod + def associated_class() -> Type[DSModuleBase]: + return DSPreNormBase diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/unembed_base.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/unembed_base.py new file mode 100644 index 0000000000000000000000000000000000000000..9eca6fcde7682903ef445f4703e009b5c1b9e641 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/interfaces/unembed_base.py @@ -0,0 +1,61 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Any, Dict, Optional, Type + +import torch + +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from ...ragged import RaggedBatchWrapper +from ..ds_module import DSModuleBase +from ..module_registry import DSModuleRegistryBase +from ..configs import DSUnembedConfig + + +class DSUnembedBase(DSModuleBase): + """ + Base mixin for unmebedding modules. The interface represented by this module is: + + if config.do_normalization + hidden = layer_norm(hidden) + logits = hidden @ projection + """ + + @staticmethod + def config_class() -> Type[DeepSpeedConfigModel]: + return DSUnembedConfig + + def __init__(self, config: DSUnembedConfig, implementation_config: Dict[str, Any]) -> None: + super().__init__(config, implementation_config) + + def forward(self, + hidden_states: torch.Tensor, + vocab_embedding: torch.Tensor, + ragged_metadata: RaggedBatchWrapper, + gamma: Optional[torch.Tensor] = None, + beta: Optional[torch.Tensor] = None) -> torch.Tensor: + """ + Forward interface. Gamma and beta are optional parameters passed depending on + `self.config.do_normalization`. + + Args: + hidden_states (torch.Tensor): Hidden states of shape [tokens, model_dim] + vocab_embedding (torch.Tensor): Embedding matrix of shape [vocab_size, model_dim] + ragged_metadata (RaggedBatchWrapper): Metadata for the ragged batch. + gamma (Optional[torch.Tensor]): Gamma parameter for layer norm. + beta (Optional[torch.Tensor]): Beta parameter for layer norm. + + Returns: + torch.Tensor: Unembedded hidden states of shape [n_seqs, model_dim] + """ + raise NotImplementedError() + + +class DSUnembedRegistry(DSModuleRegistryBase): + registry: Dict = {} + + @staticmethod + def associated_class() -> Type[DSModuleBase]: + return DSUnembedBase diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/modules/module_registry.py b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/module_registry.py new file mode 100644 index 0000000000000000000000000000000000000000..e04b8d734518b2322fe3f07a38c5d01222e4ecbe --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/modules/module_registry.py @@ -0,0 +1,58 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from abc import ABC, abstractstaticmethod +from typing import Any, Dict, Type + +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from .ds_module import DSModuleBase + + +class ConfigBundle(DeepSpeedConfigModel): + """ + A config bundle is a collection of configs that are used to instantiate a model implementation. + """ + name: str + config: DeepSpeedConfigModel + implementation_config: Dict[str, Any] = {} + + +class DSModuleRegistryBase(ABC): + """ + Class holding logic for tracking the DSModule implementations of a given interface. + """ + + @classmethod + def instantiate_config(cls, config_bundle: ConfigBundle) -> DSModuleBase: + """ + Given a DSModule key, attempt to instantiate + """ + if config_bundle.name not in cls.registry: + raise KeyError(f"Unknown DSModule: {config_bundle.name}, cls.registry={cls.registry}") + + target_implementation = cls.registry[config_bundle.name] + if not target_implementation.supports_config(config_bundle.config): + raise ValueError(f"Config {config_bundle.config} is not supported by {target_implementation}") + + return cls.registry[config_bundle.name](config_bundle.config, config_bundle.implementation_config) + + @abstractstaticmethod + def associated_class() -> Type[DSModuleBase]: + """ + Return the class associated with this registry. + """ + raise NotImplementedError("Must associated a DSModule class with its registry.") + + @classmethod + def register_module(cls, child_class: DSModuleBase) -> None: + """ + Register a module with this registry. + """ + if not issubclass(child_class, cls.associated_class()): + raise TypeError( + f"Can only register subclasses of {cls.associated_class()}, {child_class} does not inherit from {cls.associated_class()}" + ) + cls.registry[child_class.name()] = child_class + return child_class diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__init__.py b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..3af09cff4be57716e3e4295b88163db6b90b608a --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__init__.py @@ -0,0 +1,15 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from .kv_cache import split_kv +from .manager_configs import ( + AllocationMode, + DSStateManagerConfig, + KVCacheConfig, + MemoryConfig, +) +from .ragged_manager import DSStateManager +from .ragged_wrapper import RaggedBatchWrapper +from .sequence_descriptor import DSSequenceDescriptor, PlaceholderSequenceDescriptor diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ccc37b41afad8aa6574936606bf8a91d70f0fe1c Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/blocked_allocator.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/blocked_allocator.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..000e38b0687a34798e5f76a9213582c927e7b85f Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/blocked_allocator.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/kv_cache.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/kv_cache.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..0e1957eff2a7d23b02cb4d30cc3366bdb403dfeb Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/kv_cache.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/manager_configs.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/manager_configs.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f2bef74e35f9e576e2b667fad05b90fb75a38962 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/manager_configs.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/ragged_manager.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/ragged_manager.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..61157b52e891aabdbd51912f0e1fed3daebe9449 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/ragged_manager.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/ragged_wrapper.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/ragged_wrapper.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..025e2db36a3a6a0d3d53162cfb87c1061192dda2 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/ragged_wrapper.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/sequence_descriptor.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/sequence_descriptor.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..6355d6cfb231d177e8e388edf410c401508220bc Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/__pycache__/sequence_descriptor.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/blocked_allocator.py b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/blocked_allocator.py new file mode 100644 index 0000000000000000000000000000000000000000..7884d8cccb47a1e9a4d94759a12007fabfea90a4 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/blocked_allocator.py @@ -0,0 +1,105 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Iterable, Union + +import torch + + +class BlockedAllocator: + """ + Allocator class for managing which blocks are free/used in the + blocked KV-cache. This is a simple allocator that uses a linked list + to keep track of which blocks are free/used. The cost of allocation/deallocation + is O(blocks), where blocks is the number of blocks to allocate/deallocate. + + TODO(cmikeh2): Evaluate performance of this allocator and migrate + to C++ if necessary. + """ + # Number of blocks in the KV-cache(s). + _num_blocks: int + + # Array of blocks, where each element is the next block in the linked list. + _blocks: torch.Tensor + + # Index of the head of the linked list. + _head: int + + # Number of free blocks in the KV-cache. + _free_blocks: int + + def __init__(self, num_blocks: int) -> None: + """ + Initialize an allocator with `num_blocks` blocks. This requires at least + `num_blocks` * 4 bytes of host memory. + + Parameters: + num_blocks (int): The number of blocks to allocate. + """ + + if num_blocks < 1: + raise ValueError(f'Blocked KV-cache must have at least 1 block, provided {num_blocks}') + + self._num_blocks = num_blocks + self._blocks = torch.arange(1, num_blocks + 1, dtype=torch.int32, device='cpu', pin_memory=True) + self._head = 0 + self._free_blocks = num_blocks + + def allocate(self, num_blocks: int) -> torch.Tensor: + """ + Allocate a list of blocks from the associated KV-caches. This will + return `num_blocks` blocks from the KV-cache if they are available, + or raise an exception if there are not enough free blocks. + + Parameters: + num_blocks (int): The number of blocks to allocate. + + Returns: + List[int]: The list of blocks allocated. + """ + if num_blocks > self._free_blocks: + raise ValueError(f'Not enough free blocks in the KV-cache to allocate {num_blocks} blocks') + + allocated_blocks = torch.zeros(num_blocks, dtype=torch.int32) + for i in range(num_blocks): + allocated_blocks[i] = self._head + self._head = self._blocks[self._head].item() + self._blocks[allocated_blocks[i]] = -1 # Mark as used + self._free_blocks -= 1 + + return allocated_blocks + + def free(self, blocks: Union[Iterable[int], int]) -> None: + """ + Return a list of blocks to the free pool. If a single invalid block is provided (i.e., + one that is out of range of the allocator or is already free), then an exception is raised + and no blocks are freed. + + Parameters: + blocks (Union[Iterable[int], int]): The list of blocks to free. If only one block + is to be freed, this can be alone as an integer. + """ + if isinstance(blocks, int): + blocks = [blocks] + + for block in blocks: + # Parse all blocks for validity before mutating the list. + if block < 0 or block >= self._num_blocks: + raise ValueError(f'Invalid block {block} provided to free') + + if self._blocks[block] != -1: + raise ValueError(f'Block {block} is already free') + + for block in blocks: + self._blocks[block] = self._head + self._head = block + self._free_blocks += 1 + + @property + def free_blocks(self) -> int: + """ + Return the number of free blocks in the KV-cache. + """ + return self._free_blocks diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/csrc/fast_host_buffer.cu b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/csrc/fast_host_buffer.cu new file mode 100644 index 0000000000000000000000000000000000000000..31347636b50c40fc74f0c477e848a9cddbab0c92 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/csrc/fast_host_buffer.cu @@ -0,0 +1,18 @@ +// Copyright (c) Microsoft Corporation. +// SPDX-License-Identifier: Apache-2.0 + +// DeepSpeed Team + +#include "ds_kernel_utils.h" +#include "fast_host_buffer.h" + +void* get_cuda_fast_buffer(int64_t size) +{ + void* buffer_ptr; + // Host allocation flags that should minimize the host -> accelerator copy latency + unsigned int alloc_flags = + cudaHostAllocPortable | cudaHostAllocMapped | cudaHostAllocWriteCombined; + + cudaHostAlloc(&buffer_ptr, size, alloc_flags); + return buffer_ptr; +} diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/csrc/ragged_ops.cpp b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/csrc/ragged_ops.cpp new file mode 100644 index 0000000000000000000000000000000000000000..ce115f993c3c5676c77055773569b50ac68f0f7c --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/csrc/ragged_ops.cpp @@ -0,0 +1,76 @@ +// Copyright (c) Microsoft Corporation. +// SPDX-License-Identifier: Apache-2.0 + +// DeepSpeed Team + +#include +#include + +#include "fast_host_buffer.h" + +/* +Similar to doing an empty_like to replicate a Tensor on the host, but will +attempt to optimize for faster host -> accelerator copies. Since this is on the critical +path for the forward pass, this should directly improve performance. +Allocates the shadow buffers for the input_ids, batch, seq and kv_ids tensors. + +Arguments: + device_mirror: A tensor on the accelerator that should be mirrored by the host. + +Returns: + A tensor on the host of the same size and datatype optimized for fast host -> accelerator +copies. +*/ +torch::Tensor allocate_fast_host_buffer(torch::Tensor device_mirror) +{ +#ifdef __HIP_PLATFORM_AMD__ + auto options = + torch::TensorOptions().device(torch::kCPU).pinned_memory(true).dtype(device_mirror.dtype()); + auto buffer = torch::empty(device_mirror.sizes(), options); +#else + + void* buffer_ptr = get_cuda_fast_buffer(device_mirror.numel() * device_mirror.element_size()); + + auto options = torch::TensorOptions().device(torch::kCPU).dtype(device_mirror.dtype()); + auto buffer = torch::from_blob(buffer_ptr, device_mirror.sizes(), options); +#endif + return buffer; +} + +torch::Tensor allocate_view_on(torch::Tensor& tensor, torch::Tensor& buffer, int64_t offset) +{ + int8_t* data = reinterpret_cast(buffer.data_ptr()); + + auto options = tensor.options().device(buffer.device()); + + return at::from_blob(data + offset, tensor.sizes(), tensor.strides(), options); +} + +torch::Tensor allocate_view_like(py::tuple shape, + py::tuple strides, + torch::Tensor& dummy_tensor, + torch::Tensor& buffer, + int64_t offset) +{ + int8_t* data = reinterpret_cast(buffer.data_ptr()); + + auto options = torch::TensorOptions().device(buffer.device()).dtype(dummy_tensor.dtype()); + + return at::from_blob(data + offset, + shape.cast>(), + strides.cast>(), + options); +} + +PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) +{ + m.def("allocate_fast_host_buffer", + &allocate_fast_host_buffer, + "Allocate a host mirror of an accelerator Tensor."); + m.def("allocate_view_on", + &allocate_view_on, + "Allocate a view on a Tensor on the same device as the input Tensor."); + m.def("allocate_view_like", + &allocate_view_like, + "Allocate a view on a Tensor on the same device as the input Tensor."); +} diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/includes/fast_host_buffer.h b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/includes/fast_host_buffer.h new file mode 100644 index 0000000000000000000000000000000000000000..81f24ed8fdaadfcf1b712e7c025e6f7a2e4a95a1 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/includes/fast_host_buffer.h @@ -0,0 +1,14 @@ +// Copyright (c) Microsoft Corporation. +// SPDX-License-Identifier: Apache-2.0 + +// DeepSpeed Team + +#pragma once + +#include "ds_kernel_utils.h" + +/* +Wrapper around cudaHostAlloc with some specific flags. Returns a pointer to the +memory region of `size` bytes. +*/ +void* get_cuda_fast_buffer(int64_t size); diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/kv_cache.py b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/kv_cache.py new file mode 100644 index 0000000000000000000000000000000000000000..ceba3190b93cdc378bc4581def235ffa99210bfd --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/kv_cache.py @@ -0,0 +1,208 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import operator +from functools import reduce +from typing import Any, Iterable, Optional, Tuple + +import torch + +import deepspeed.comm as dist +from deepspeed.comm.reduce_op import ReduceOp + +from deepspeed.accelerator import get_accelerator +from ..inference_utils import elem_size +from ..logging import inference_logger +from .blocked_allocator import BlockedAllocator +from .manager_configs import AllocationMode, KVCacheConfig, MemoryConfig + + +def split_kv(kv_cache: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]: + """ + Split a KV cache instance into its key and value components. + + Parameters: + kv_cache (torch.Tensor): The KV-cache to split. This should be a 5D tensor with the + following shape: [num_blocks, block_size, 2, num_heads, head_size] + + Returns: + Tuple[torch.Tensor, torch.Tensor]: The key and value components of the KV-cache. Both + tensors will have the shape [num_blocks, block_size, num_heads, head_size]. + """ + if kv_cache.ndim != 5: + raise ValueError(f"KV-cache must have 5 dimensions, got {kv_cache.ndim}.") + + return kv_cache[:, :, 0, :, :], kv_cache[:, :, 1, :, :] + + +class BlockedKVCache: + + _caches: Tuple[torch.Tensor, ...] + """ + Backing storage for all KV caches. This is a 6D tensor with the following shape: + (num_caches, num_blocks, block_size, 2, num_heads, head_size) + """ + + _allocators: Tuple[BlockedAllocator, ...] + """ + Block allocator for tracking cache usage. This manages the GPU cache. + """ + + _configs: Tuple[KVCacheConfig, ...] + """ + Configuration of the KV cache(s). See ``KVCacheConfig`` for more details. This enables the support + for different types/shapes of KV-caches (i.e. the alternating local and global attention in + GPT-Neo). + """ + + def __init__(self, + configs: Tuple[KVCacheConfig, ...], + memory_config: MemoryConfig, + mp_group: Optional[Any] = None, + offload: bool = False) -> None: + """ + Create a container that will maintain the storage and allocations for a set of + blocked KV-caches. + + Parameters: + config (KVCacheConfig): The configuration of the KV-cache. + slack (int): The amount of slack space to reserve in GPU memory for the cache. + enable_offload (bool): Whether to enable offloading of the cache to the host. + blocks (int): The number of blocks to pre-allocate for the cache. If this is set, + slack will be ignored. + """ + self._configs = configs + self._memory_config = memory_config + self._enable_offload = offload + + if self._enable_offload: + raise NotImplementedError("Offloading of KV-caches is not yet supported.") + + if AllocationMode(self._memory_config.mode) is AllocationMode.RESERVE: + # TODO(cmikeh2): Change the weighting based on the type of the KV-cache + + total_per_block_footprint = 0 + for config in self._configs: + per_block_footprint = reduce(operator.mul, config.cache_shape, config.block_size) + per_block_footprint *= 2 # for key and value + total_per_block_footprint += per_block_footprint * elem_size(config.cache_dtype) + + # Perform a dummy nccl call before calculating available memory, on some systems (H100) we've observed higher memory allocations from NCCL + if dist.get_world_size(group=mp_group) > 1: + dummy_tensor = torch.tensor(0, dtype=torch.int32, device=get_accelerator().current_device()) + dist.all_reduce(dummy_tensor, op=ReduceOp.MIN, group=mp_group) + + get_accelerator().empty_cache() + available_kv_memory = get_accelerator().available_memory() - self._memory_config.size + total_memory = get_accelerator().total_memory() + + inference_logger().debug( + f"Memory usage before KV-cache allocation: total_memory={total_memory}, available_kv_memory={available_kv_memory}, total_per_block_footprint={total_per_block_footprint}" + ) + + if available_kv_memory < total_per_block_footprint: + raise ValueError( + f"Insufficient memory to allocate KV-caches. Required: {total_per_block_footprint}, Available: {available_kv_memory}" + ) + + num_blocks = available_kv_memory // total_per_block_footprint + + # In a multi-process setting, we need to ensure that all processes have the same + # KV cache capacity to ensure scheduling guarantees are equivalent on all ranks. + if dist.get_world_size(group=mp_group) > 1: + reduce_tensor = torch.tensor(num_blocks, dtype=torch.int32, device=get_accelerator().current_device()) + dist.all_reduce(reduce_tensor, op=ReduceOp.MIN, group=mp_group) + num_blocks = reduce_tensor.item() + + # This is ugly but don't want the fragmentation of the 8 byte Tensor maybe + # hanging around. + del reduce_tensor + get_accelerator().empty_cache() + else: # AllocationMode.ALLOCATE + num_blocks = self._memory_config.size + + caches = [] + allocators = [] + + for cache_group_id, config in enumerate(self._configs): + num_caches = config.cache_shape[0] + num_heads = config.cache_shape[1] + head_size = config.cache_shape[2] + + alloc_shape = (num_caches, num_blocks, config.block_size, 2, num_heads, head_size) + inference_logger().info( + f"Allocating KV-cache {cache_group_id} with shape: {alloc_shape} consisting of {num_blocks} blocks.") + caches.append(torch.empty(alloc_shape, dtype=config.cache_dtype, + device=get_accelerator().current_device())) + allocators.append(BlockedAllocator(num_blocks)) + + self._caches = tuple(caches) + self._allocators = tuple(allocators) + + def reserve(self, num_blocks: int, cache_group: int = 0) -> torch.Tensor: + """ + Reserve a number of blocks from the cache. This will return a 1D tensor of + block_ids that have been marked as reserved. + + Parameters: + num_blocks (int): The number of blocks to reserve. + cache_group (int): The cache group to reserve from. Default is 0. + """ + return self._allocators[cache_group].allocate(num_blocks) + + def free(self, blocks: Iterable[int], cache_group: int = 0) -> None: + """ + Free a set of blocks from the cache. This will mark the blocks as free in the + allocator. + + Parameters: + blocks (Iterable[int]): The blocks to free. + cache_group (int): The cache group to free from. Default is 0. + """ + self._allocators[cache_group].free(blocks) + + def offload(self, blocks: Iterable[int], cache_group: int = 0) -> torch.Tensor: + """ + Offload KV-cache blocks from accelerator memory to the host. + + Parameters: + blocks (Iterable[int]): The blocks to offload. + cache_group (int): The cache group to offload from. Default is 0. + """ + raise NotImplementedError("Offloading is not yet supported.") + + def restore(self, blocks: Iterable[int], cache_group: int = 0) -> torch.Tensor: + """ + Restore KV-cache blocks from the host to accelerator memory. + + Parameters: + blocks (Iterable[int]): The blocks to restore. + cache_group (int): The cache group to restore to. Default is 0. + """ + raise NotImplementedError("Offloading is not yet supported.") + + def get_cache(self, cache_id: int, cache_group: int = 0) -> torch.Tensor: + """ + Get the tensor associated with the given cache ID. + + Parameters: + cache_id (int): The ID of the cache tensor to get. + cache_group (int): The cache group to get from. Default is 0. + """ + return self._caches[cache_group][cache_id] + + @property + def free_blocks(self) -> torch.Tensor: + """ + Return the number of free blocks in each cache + """ + return [allocator.free_blocks for allocator in self._allocators] + + @property + def num_caches(self) -> int: + """ + Return the number of caches + """ + return len(self._caches) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/manager_configs.py b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/manager_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..17283b8bc0c4707ed95ad5e91414d160b59374ea --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/manager_configs.py @@ -0,0 +1,181 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from enum import Enum +from typing import Tuple + +from pydantic import PositiveInt, model_validator + +from deepspeed.runtime.config_utils import DeepSpeedConfigModel +from ..inference_utils import DtypeEnum + + +class KVCacheType(Enum): + + DENSE = "dense" + """ + Dense KV-cache. This is the default type. + """ + + LOCAL = "local" + """ + KV-cache that attends to only a local (trailing) window of tokens. + """ + + +class KVCacheConfig(DeepSpeedConfigModel): + + type: KVCacheType = KVCacheType.DENSE + """ + Type of KV-cache to use. This may inform the allocator of the expected access/retention pattern + to enable more efficient memory management. + """ + + block_size: int = 128 + """ + Number of tokens that may be contained in each cache block. + """ + + num_allocation_groups: PositiveInt = 1 + """ + Allocation groups are assumed to be able to use the same allocation block size because + the allocation granularity is the same but the number of blocks required in each group + may differ. + + As a concrete example, consider a model with alternating layers of local and global + attention (such as GPTNeo). The local attention layers do not require the same number + of cache blocks as the global layer. However, a static partitioning scheme is sub-optimal since the ratio of local to global KV-cache blocks is not constant across + the range of sequence lengths that may be encountered. + + NOTE: In theory, this functionality could be used to do per-head and per-layer + KV-cache allocation, but it is likely the allocator will struggle with managing that + many blocks. + + NOTE: This will need to be primarily understood and handled by the model implementation + itself, rather than the KV cache manager. However, I'd like to make this explicit. + """ + + cache_shape: Tuple[PositiveInt, PositiveInt, PositiveInt] + """ + The shape of the cache per token. The first dimension is the number of individual + caches, the second is the number of heads, and the third is the head size. The number + of caches argument here is per allocation group. + """ + + cache_dtype: DtypeEnum = DtypeEnum.fp16 + """ + Data type of the KV-cache. + """ + + max_blocks_per_allocation_group: PositiveInt = 64 + """ + Maximum number of blocks that can be associated with an allocation group. + """ + + +""" +The config above is a little confusing so let's use a couple of concrete examples of +usage: + +Model 1: Llama-13B with a block size of 256 + +Llama is uniform attention so we have a single allocation group. The cache shape is +(40 layers, 40 heads, 128 head size) + +```python +llama_kv_config = KVCacheConfig(block_size=256, + num_allocation_groups=1, + cache_shape=(40, 40, 128)) +``` + +Model 2: GPTNeo-2.7B with a block size of 128 + +GPTNeo has alternating local and global attention layers. We have two allocation groups. +There are 16 layers of each type with 20 heads apiece at 128 head size. + +```python +gptneo_kv_config = KVCacheConfig(num_allocation_groups=2, cache_shape=(16, 20, 128)) +``` +""" + + +class AllocationMode(Enum): + """ + Helper class to describe memory allocation strategies for the KV-cache. + """ + + RESERVE = "reserve" + """ + Reserve a small amount of memory for non-KV cache allocations. + """ + + ALLOCATE = "allocate" + """ + Allocate an explicit number of KV blocks. + """ + + +class MemoryConfig(DeepSpeedConfigModel): + + mode: AllocationMode = AllocationMode.RESERVE + + size: PositiveInt = 1_000_000_000 + """ + Parameter for each of the modes. + + If mode is RESERVE, this is the amount of memory in bytes to reserve after allocating the + KV-cache. If in a tensor-parallel regime, this amount is guaranteed to be reserved on + all devices. + + If mode is ALLOCATE, this is the number of blocks to allocate for the KV-cache. This may + require tuning for model/GPU setups. + """ + + +class DSStateManagerConfig(DeepSpeedConfigModel): + + max_tracked_sequences: PositiveInt = 2048 + """ + How many sequences this engine will track simultaneously. This limit should be greater + than the ``max_ragged_sequence_count``. + """ + + max_ragged_batch_size: PositiveInt = 768 + """ + The maximum number of tokens that can be contained in a single ragged batch. Passing + a larger value than this will raise an exception that must be handled by the runtime. + """ + + max_ragged_sequence_count: PositiveInt = 512 + """ + The maximum number of sequences that can compose a batch. This limitation is only + relevant under CUDA graphing scenarios currently, where the maximum number of blocks + is largely bound by the total number of sequences in the ragged batch. This number cannot + be larger than ``max_tracked_sequences`` or ``max_ragged_batch_size``. + """ + + max_context: PositiveInt = 8192 + """ + The maximum number of tokens (inclusive of generation) that can be contained in a single + sequence. Currently used to bound the size of the KV cache metadata. + """ + + memory_config: MemoryConfig = MemoryConfig() + """ + Directive for how to manage the creation of the KV-cache. See MemoryConfig for more + details. + """ + + offload: bool = False + """ + Enable tracking for offloading KV-cache to host memory. Currently unsupported. + """ + + @model_validator(mode="after") + def max_ragged_sequence_count_validator(self): + # If the attributes below failed their validation they won't appear in the values dict. + assert self.max_ragged_sequence_count <= self.max_tracked_sequences, "max_ragged_sequence_count must be less than max_tracked_sequences" + assert self.max_ragged_sequence_count <= self.max_ragged_batch_size, "max_ragged_sequence_count must be less than max_ragged_batch_size" + return self diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/ragged_manager.py b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/ragged_manager.py new file mode 100644 index 0000000000000000000000000000000000000000..ecc3c52a5834321255d3e475fc2fcf8eb391dd82 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/ragged_manager.py @@ -0,0 +1,206 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import torch +from typing import Any, Dict, Optional, Tuple + +from deepspeed.accelerator import get_accelerator +from deepspeed.ops.op_builder import RaggedUtilsBuilder +from deepspeed.utils.logging import logger + +from .blocked_allocator import BlockedAllocator +from .kv_cache import BlockedKVCache +from .manager_configs import DSStateManagerConfig, KVCacheConfig +from .sequence_descriptor import DSSequenceDescriptor + + +class DSStateManager: + """ + Base abstract class for managing blocked KV caches. Will probably have a single + implementation for now. + """ + + _config: DSStateManagerConfig + """ + Config for state management. See DSStateManagerConfig for more details. The arguments here + should come from the engine config. + """ + + _kv_configs: Tuple[KVCacheConfig] + """ + Config for the KV cache. See KVCacheConfig for more details. These arguments should derive + from the model implementation. + """ + + _kv_cache: BlockedKVCache + """ + Persistent KV cache store. + """ + + # Container for tracking all sequences in the system. + _seqs: Dict[int, DSSequenceDescriptor] + """ + Container for tracking all sequences in the system. + + TODO(cmikeh2): Evaluate if this has any performance implications. + """ + + # Allocator for tracking sequences. + _tracking_allocator: BlockedAllocator + _all_block_ids: Tuple[torch.Tensor, ...] + _all_block_ids_shadow: Tuple[torch.Tensor, ...] + + def __init__(self, + config: DSStateManagerConfig, + kv_configs: Tuple[KVCacheConfig, ...], + base_mp_group: Optional[Any] = None) -> None: + """ + The key + + Parameters: + block_size (int): The number of tokens to allocate in each block. + """ + self._config = config + self._kv_configs = kv_configs + + # Load our helpers for host allocation. + self._ragged_utils = RaggedUtilsBuilder().load() + + # Initialize the allocator for tracking sequences (so this doesn't need to be ad-hoc). + self._tracking_allocator = BlockedAllocator(self._config.max_tracked_sequences) + + all_block_ids = [] + all_block_ids_shadow = [] + + for cache_config in self._kv_configs: + # Storage to back tracking the KV cache allocation. + ids_shape = ( + self._config.max_tracked_sequences, + cache_config.num_allocation_groups, + cache_config.max_blocks_per_allocation_group, + ) + + all_block_ids.append(torch.zeros(ids_shape, dtype=torch.int32, device=get_accelerator().current_device())) + all_block_ids_shadow.append(self._ragged_utils.allocate_fast_host_buffer(all_block_ids[-1])) + + self._all_block_ids = tuple(all_block_ids) + self._all_block_ids_shadow = tuple(all_block_ids_shadow) + + # Initialize the sequence container. + self._seqs = {} + + # Finally initialize the KV cache. + self._kv_cache = BlockedKVCache(self._kv_configs, + self._config.memory_config, + mp_group=base_mp_group, + offload=self._config.offload) + + def get_cache(self, cache_id: int, cache_group: int = 0) -> torch.Tensor: + """ + Return the Tensor associated with the given cache id in the specified cache group. + + Arguments: + cache_group (str): The KV cache group. + cache_id (int): The cache id within that group. + """ + return self._kv_cache.get_cache(cache_id, cache_group=cache_group) + + def flush_sequence(self, uid: int) -> None: + """ + Free all resources associated with the given sequence id. + """ + if uid not in self._seqs: + logger.warning(f"Attempting to flush sequence {uid} which does not exist.") + return + + seq = self._seqs[uid] + for i in range(self.n_kv_cache_groups): + self._kv_cache.free(seq.all_block_ids(cache_group=i), cache_group=i) + + self._tracking_allocator.free(seq.tracking_id) + del self._seqs[uid] + + def get_sequence(self, uid: int) -> Optional[DSSequenceDescriptor]: + """ + Get the sequence descriptor for the given sequence id. If the sequence does not exist, + then None is returned. + """ + return self._seqs.get(uid, None) + + def get_or_create_sequence(self, uid: int) -> DSSequenceDescriptor: + """ + Get the existing sequence descriptor for a given uid or initialize one if + it does not exist. NOTE: This will always return a valid sequence descriptor + if one may be allocated and should not be used from APIs that are attempting + to test the schedulability of a hypothetical batch. + """ + seq = self.get_sequence(uid) + if seq is not None: + return seq + else: + return self._create_sequence(uid) + + def _create_sequence(self, uid: int) -> DSSequenceDescriptor: + """ + Create a new sequence descriptor for the given sequence id. + """ + if uid in self._seqs: + raise ValueError(f"Sequence {uid} already exists.") + + try: + tracking_slot = self._tracking_allocator.allocate(1).item() + except ValueError: + raise RuntimeError( + f"Unable to create tracking slot for sequence {uid} since the metadata buffers are full.") + + seq_block_ids = tuple(all_block_ids[tracking_slot] for all_block_ids in self._all_block_ids) + seq_block_ids_shadow = tuple(all_block_ids_shadow[tracking_slot] + for all_block_ids_shadow in self._all_block_ids_shadow) + + self._seqs[uid] = DSSequenceDescriptor(tracking_slot, + seq_block_ids, + seq_block_ids_shadow, + max_context=self._config.max_context) + # TODO(cmikeh2): Debug call here might be unnecessary and is potentially on critical path. + logger.debug(f"Created sequence {uid} with tracking slot {tracking_slot}.") + return self._seqs[uid] + + @property + def tracked_sequences(self) -> Dict[int, DSSequenceDescriptor]: + """ + Return the tracked sequences. + """ + return self._seqs + + @property + def n_tracked_sequences(self) -> int: + """ + Return the number of sequences currently tracked. + """ + return len(self._seqs) + + @property + def kv_block_size(self) -> int: + """ + Return the block size of the KV cache. + """ + return self._kv_config.block_size + + @property + def n_kv_cache_groups(self) -> int: + """ + Return the number of KV caches. + """ + return self._kv_cache.num_caches + + @property + def free_blocks(self) -> torch.Tensor: + """ + Return the number of free blocks in the KV cache. + """ + return self._kv_cache.free_blocks + + def allocate_blocks(self, n_blocks: int, cache_group: int = 0) -> torch.Tensor: + return self._kv_cache.reserve(n_blocks, cache_group=cache_group) diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/ragged_wrapper.py b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/ragged_wrapper.py new file mode 100644 index 0000000000000000000000000000000000000000..056ecfa2ac4094cf9380db533a1f031332815641 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/ragged_wrapper.py @@ -0,0 +1,292 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Optional + +import torch + +from deepspeed.accelerator import get_accelerator +from deepspeed.ops.op_builder import RaggedUtilsBuilder + +from .sequence_descriptor import DSSequenceDescriptor +from .manager_configs import DSStateManagerConfig + + +def to_padded(original_size: int) -> int: + """ + Pad to a backend friendly granularity. + """ + + def _pad_to_mul_of_pow2(val: int, pow_2_val: int) -> int: + return val + (pow_2_val - 1) & ~(pow_2_val - 1) + + # TODO(cmikeh2): Tune this approach. This is mainly a placeholder right now. + granularity = 64 if original_size <= 512 else 128 + + return _pad_to_mul_of_pow2(original_size, granularity) + + +class RaggedBatchWrapper: + """ + Container for all the auxiliary Tensors used in the management of a ragged batch. + + For each Tensor, we maintain a shadow Tensor on the host. This Tensor is what is + directly populated when constructing the ragged batch. The shadow Tensors, when possible, + should be allocated so as to support fast host-to-accelerator copies. + """ + + # Tensors to populate the ragged batch into. + _input_ids_shadow: torch.Tensor + _input_ids: torch.Tensor + """ + Forward pass input buffer. + """ + + _batch_metadata_storage: torch.Tensor + _batch_metadata_storage_shadow: torch.Tensor + """ + Holds the number of inflight sequences and tokens for the ragged batch. + """ + + _token_to_seq_storage: torch.Tensor + _token_to_seq_storage_shadow: torch.Tensor + """ + Linear mapping for each of the tokens. Let's say we have 8 tokens in the batch, + with the sequence breakdown being [4, 1, 3]. Then, the mapping would be: + [0, 0, 0, 0, 1, 2, 2, 2] + """ + + _inflight_seq_descriptors: torch.Tensor + _inflight_seq_descriptors_shadow: torch.Tensor + """ + For each sequence in the batch, we store the start token in the batch, the number of tokens + the number of tokens in the history of this sequence, and an unused 4th reserved for alignment. + For the above example this would give: + [[0, 4, H0, X], [4, 1, H1, X], [5, 3, H2, X]] + """ + + # Holds the block ids for each sequence in the ragged batch. + _kv_ptrs: torch.Tensor + _kv_ptrs_shadow: torch.Tensor + """ + List of ptrs pointing to the GPU buffer that holds the KV-block ids for each sequence. + If there are multiple allocation groups associated with each of the sequences, then + then accessing the Nth cache will require accessing the Nth block id + """ + + def __init__(self, config: DSStateManagerConfig) -> None: + """ + Convenience wrapper around the data structures used to represent a ragged + batch for inference. Only a single `RaggedBatchWrapper` should be used per + ragged inference engine. + + The underlying data structures are implemented in `ragged_batch_descriptor.h`. + """ + self._config = config + self._input_ids = torch.zeros((self._config.max_ragged_batch_size), + dtype=torch.int64, + device=get_accelerator().current_device()) + + self._batch_metadata_storage = torch.zeros(2, dtype=torch.int32, device=get_accelerator().current_device()) + + self._token_to_seq_storage = torch.zeros((self._config.max_ragged_batch_size), + dtype=torch.int32, + device=get_accelerator().current_device()) + self._inflight_seq_descriptors = torch.zeros((self._config.max_ragged_sequence_count, 4), + dtype=torch.int32, + device=get_accelerator().current_device()) + self._kv_ptrs = torch.zeros((self._config.max_ragged_sequence_count), + dtype=torch.int64, + device=get_accelerator().current_device()) + + self._utils_module = RaggedUtilsBuilder().load() + host_alloc = self._utils_module.allocate_fast_host_buffer + + self._input_ids_shadow = host_alloc(self._input_ids) + self._batch_metadata_storage_shadow = host_alloc(self._batch_metadata_storage) + self._token_to_seq_storage_shadow = host_alloc(self._token_to_seq_storage) + self._inflight_seq_descriptors_shadow = host_alloc(self._inflight_seq_descriptors) + self._kv_ptrs_shadow = host_alloc(self._kv_ptrs) + + # Default behavior should be no padding + self._is_padded = False + + self._current_tokens = 0 + self._current_sequences = 0 + self._batch_tokens = [] + self._inflight_seq_descriptors_shadow_buf = [] + self._kv_blocks_ptr_buf = [] + self._token_to_seq_storage_shadow_buf = [] + + def clear(self) -> None: + """ + Clear the ragged batch. This will reset the number of tokens and sequences to 0. + """ + self._current_tokens = 0 + self._current_sequences = 0 + self._batch_tokens = [] + self._inflight_seq_descriptors_shadow_buf = [] + self._kv_blocks_ptr_buf = [] + self._token_to_seq_storage_shadow_buf = [] + + def insert_sequence(self, seq_descriptor: DSSequenceDescriptor, tokens: torch.Tensor, do_checks=True) -> None: + """ + Incrementally insert a sequence into the ragged batch. This will update the + metadata for the ragged batch and the sequence. + + Arguments: + seq_descriptor () + """ + if tokens.device != torch.device("cpu"): + # This doesn't really fall under schedulability, so we'll unconditionally check for it. + raise RuntimeError(f"Expected tokens to be on host but found device '{tokens.device}'") + + if do_checks and self.current_sequences == self._config.max_ragged_sequence_count: + raise RuntimeError(f"Ragged batch is full due to sequence limit: {self._config.max_ragged_sequence_count}") + + seq_tokens = tokens.numel() + + if do_checks and self.current_tokens + seq_tokens > self._config.max_ragged_batch_size: + raise RuntimeError(f"Ragged batch is full due to capacity limit: {self._config.max_ragged_batch_size})") + + # The values in _inflight_seq_descriptors_shadow_buf, _token_to_seq_storage_shadow_buf, _kv_blocks_ptr_buf, etc., + # are ultimately stored in PyTorch tensors: _inflight_seq_descriptors_shadow, _token_to_seq_storage_shadow, _kv_ptrs_shadow, etc. + # However, we found it inefficient to iterate over and substitute values into tensor slices or to use copy/fill calls for this purpose. + # Therefore, we initially store the values in Python lists or primitive data types and then copy them collectively in the finalize() method, + # instead of updating the tensors directly in each iteration. + self._batch_tokens.append(tokens) + self._inflight_seq_descriptors_shadow_buf.append(self.current_tokens) + self._inflight_seq_descriptors_shadow_buf.append(seq_tokens) + self._inflight_seq_descriptors_shadow_buf.append(seq_descriptor.seen_tokens) + self._inflight_seq_descriptors_shadow_buf.append(0) # alignment + + self._token_to_seq_storage_shadow_buf.extend([self.current_sequences] * seq_tokens) + + self._kv_blocks_ptr_buf.append(seq_descriptor.kv_blocks_ptr) + + self._current_tokens += seq_tokens + self._current_sequences += 1 + + @property + def tensor_toks(self) -> torch.Tensor: + """ + The number of tokens in the in-flight ragged batch. This will not trigger + synchronization with the device. + """ + cur_toks = self.current_tokens + if self._is_padded: + return to_padded(cur_toks) + else: + return cur_toks + + def finalize(self, padding: Optional[bool] = False) -> None: + """ + Completes construction of the ragged batch by flushing the host buffers to the device. + """ + cur_toks = self.current_tokens + + # Batch-copy the values recorded in insert_sequence() into PyTorch tensors to enhance efficiency. + self._inflight_seq_descriptors_shadow.flatten()[:len(self._inflight_seq_descriptors_shadow_buf)].copy_( + torch.tensor(self._inflight_seq_descriptors_shadow_buf)) + self._input_ids_shadow[:self.current_tokens].copy_(torch.cat(self._batch_tokens, dim=0)) + self._token_to_seq_storage_shadow[:len(self._token_to_seq_storage_shadow_buf)].copy_( + torch.tensor(self._token_to_seq_storage_shadow_buf)) + self._kv_ptrs_shadow[:len(self._kv_blocks_ptr_buf)].copy_(torch.tensor(self._kv_blocks_ptr_buf)) + self._batch_metadata_storage_shadow.copy_(torch.tensor([cur_toks, self.current_sequences])) + + if padding: + padded_toks = to_padded(cur_toks) + self._input_ids_shadow[cur_toks:padded_toks].fill_(-1) + self._token_to_seq_storage_shadow[cur_toks:padded_toks].fill_(-1) + self._is_padded = True + else: + padded_toks = cur_toks + self._is_padded = False + + current_sequences = self.current_sequences + + def _noblock_copy(dst: torch.Tensor, src: torch.Tensor) -> None: + dst.copy_(src, non_blocking=True) + + _noblock_copy(self._input_ids[:padded_toks], self._input_ids_shadow[:padded_toks]) + _noblock_copy(self._batch_metadata_storage, self._batch_metadata_storage_shadow) + _noblock_copy(self._token_to_seq_storage[:padded_toks], self._token_to_seq_storage_shadow[:padded_toks]) + _noblock_copy(self._inflight_seq_descriptors[:current_sequences], + self._inflight_seq_descriptors_shadow[:current_sequences]) + _noblock_copy(self._kv_ptrs[:current_sequences], self._kv_ptrs_shadow[:current_sequences]) + + def input_ids(self, on_device: bool = True) -> torch.Tensor: + """ + The input ids tensor for the ragged batch. If the device Tensor is requested, the Tensor + is truncated to the number of tokens in the batch. + """ + if on_device: + return self._input_ids[:self.tensor_toks] + else: + return self._input_ids_shadow + + def batch_metadata_buffer(self, on_device: bool = True) -> torch.Tensor: + """ + Buffer associated with the batch metadata tensor that can + be populated in preparation for passing a new input to the device. + """ + if on_device: + return self._batch_metadata_storage + else: + return self._batch_metadata_storage_shadow + + def tokens_to_seq(self, on_device: bool = True) -> torch.Tensor: + """ + Mapping of token to which sequence it belongs to in the ragged batch. If the device Tensor + is requested, the Tensor is truncated to the number of tokens in the batch. + """ + if on_device: + return self._token_to_seq_storage[:self.tensor_toks] + else: + return self._token_to_seq_storage_shadow + + def inflight_seq_descriptors(self, on_device: bool = True) -> torch.Tensor: + """ + Buffer associated with the metadata of each sequence in the ragged batch. If the device Tensor + is requested, the Tensor is truncated to the number of sequences in the batch. + """ + if on_device: + return self._inflight_seq_descriptors[:self.current_sequences] + else: + return self._inflight_seq_descriptors_shadow + + def kv_ptrs(self, on_device: bool = True) -> torch.Tensor: + """ + Pointer to where the list of KV ids associated with a sequence are. If the device Tensor + is requested, the Tensor is truncated to the number of sequences in the batch. + """ + if on_device: + return self._kv_ptrs[:self.current_sequences] + else: + return self._kv_ptrs_shadow + + def masks(self, on_device: bool = True) -> Optional[torch.Tensor]: + """ + Placeholder for supporting complex masks. Currently not supported. + + Models that will need this will be BERT-like, not generative. + """ + return None + + @property + def current_tokens(self) -> int: + """ + The number of tokens in the in-flight ragged batch. This will not trigger + synchronization with the device. + """ + return self._current_tokens + + @property + def current_sequences(self) -> int: + """ + The number of sequences in the in-flight ragged batch. This will not trigger + synchronization with the device. + """ + return self._current_sequences diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/sequence_descriptor.py b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/sequence_descriptor.py new file mode 100644 index 0000000000000000000000000000000000000000..6b9f65255eec8ed0996c45e224941779777ef7f3 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/ragged/sequence_descriptor.py @@ -0,0 +1,280 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import List, Tuple, Union + +import torch + + +class BaseSequenceDescriptor: + + @property + def seen_tokens(self) -> int: + """ + The number of tokens for this sequence that have completed a forward pass. + """ + raise NotImplementedError() + + @property + def cur_allocated_blocks(self, cache_group: int = 0) -> int: + """ + The number of KV blocks currently allocated for this sequence. + """ + raise NotImplementedError() + + @property + def kv_blocks_ptr(self, cache_group: int = 0) -> int: + """ + The pointer to the KV blocks for this sequence. + """ + raise NotImplementedError() + + +class PlaceholderSequenceDescriptor(BaseSequenceDescriptor): + """ + The DummySequenceDescriptor is an empty object that allows us to perform schedulability + checks before formally tracking a sequence. + """ + + def __init__(self, seen_tokens=0, cur_allocated_blocks=0, kv_blocks_ptr=0) -> None: + self._seen_tokens = seen_tokens + self._cur_allocated_blocks = cur_allocated_blocks + self._kv_blocks_ptr = kv_blocks_ptr + + @property + def seen_tokens(self) -> int: + return self._seen_tokens + + @property + def cur_allocated_blocks(self, cache_group: int = 0) -> int: + return self._cur_allocated_blocks + + @property + def kv_blocks_ptr(self, cache_group: int = 0) -> int: + return self._kv_blocks_ptr + + +class DSSequenceDescriptor(BaseSequenceDescriptor): + + _seen_tokens: int + """ + Number of tokens in the sequence that have completed a forward pass. + """ + + _in_flight_tokens: int + """ + Number of tokens that have begun a forward pass but not yet completed it. + """ + + _max_context: int + """ + Maximum number of tokens this sequence may eventually include. Currently unused but + may be used in future implementations for speculative caching. + """ + + _num_allocation_groups: Tuple[int, ...] + """ + Number of unique allocation groups associated with the sequence for each cache group. + """ + + _blocks_per_allocation_group: Tuple[torch.IntTensor, ...] + """ + Number of blocks allocated for each allocation group in each cache group. + """ + + # Padded list of KV-cache IDs for the sequence. + _kv_cache_ids: Tuple[torch.Tensor, ...] + _kv_cache_ids_shadow: Tuple[torch.Tensor, ...] + """ + Padded list of KV-cache IDs for the sequence. The padded shape is [num_allocation_groups, max_blocks_per_allocation_group]. + """ + + # The location in the broader ID tensor where the KV-cache IDs for the sequence + # are stored. Used on flush. + _tracking_id: int + + def __init__(self, + tracking_id: int, + kv_cache_ids: Tuple[torch.Tensor, ...], + kv_cache_ids_shadow: Tuple[torch.Tensor, ...], + max_context: int = -1) -> None: + """ + Create the metadata to track a single sequence in the system. + + Arguments: + tracking_id (int): The slot in the tracking buffers used to track this sequence. + kv_cache_ids (Tuple[torch.Tensor, ...]): The KV-cache IDs for the sequence. The shape + of the tensor should be [num_allocation_groups, max_blocks_per_allocation_group]. + There should be one tensor per cache group. + kv_cache_ids_shadow (Tuple[torch.Tensor, ...]): The shadow tensor for the KV-cache IDs. + This tensor should be allocated on the host and should have the same shape as the + tensor provided in ``kv_cache_ids``. There should be one tensor per cache group. + max_context (int): The maximum number of tokens this sequence may eventually include. + Currently unused but may be used in future implementations for speculative caching. + """ + self._tracking_id = tracking_id + self._kv_cache_ids = kv_cache_ids + self._kv_cache_ids_shadow = kv_cache_ids_shadow + self._max_context = max_context + self._n_cache_groups = len(kv_cache_ids) + + self._seen_tokens = 0 + self._in_flight_tokens = 0 + + self._num_allocation_groups = tuple(kv_cache_ids_shadow.shape[0] + for kv_cache_ids_shadow in kv_cache_ids_shadow) + self._blocks_per_allocation_group = tuple( + torch.zeros(num_groups, dtype=torch.int32, device="cpu") for num_groups in self._num_allocation_groups) + + for cache_group, kv_cache_ids in enumerate(kv_cache_ids): + assert self._num_allocation_groups[cache_group] == kv_cache_ids.shape[0] + assert len(kv_cache_ids.shape) == 2 + + @property + def seen_tokens(self) -> int: + """ + Number of tokens in the sequence that have completed a forward pass. + """ + return self._seen_tokens + + @property + def in_flight_tokens(self) -> int: + """ + Number of tokens that have begun a forward pass but not yet completed it. + """ + return self._in_flight_tokens + + @property + def max_context(self) -> int: + """ + Maximum number of tokens for this sequence. Currently unused. + """ + return self._max_context + + @property + def tracking_id(self) -> int: + """ + Return the slot in the tracking buffers used to track this sequence. + """ + return self._tracking_id + + @property + def cur_allocated_blocks(self, cache_group: int = 0) -> int: + """ + Returns the number of blocks currently allocated for this sequence in the specified cache group. + + Arguments: + cache_group (int): The cache group to query. + """ + # Currently, there is only one allocation group. + # A shortcut is used here to bypass the overhead of sum(). + if len(self._blocks_per_allocation_group) == 1: + return self._blocks_per_allocation_group[0].item() + return self._blocks_per_allocation_group[cache_group].sum().item() + + def kv_cache_ids(self, cache_group: int = 0, on_device: bool = False) -> torch.Tensor: + """ + Returns the Tensor containing the block IDs for this sequence on the appropriate device + for the specified cache group. + + Arguments: + cache_group (int): The cache group to query. + on_device (bool): Whether or not to return the Tensor on the device or on the host. + """ + if on_device: + return self._kv_cache_ids[cache_group] + else: + return self._kv_cache_ids_shadow[cache_group] + + @property + def kv_blocks_ptr(self, cache_group: int = 0) -> int: + """ + Get the device pointer to the base of the KV-cache ids for the specified cache group and + sequence. + + Arguments: + cache_group (int): The cache group to query. + """ + return self._kv_cache_ids[cache_group].data_ptr() + + #TODO: this was previously a property but causing issues with PR-4668 need to consult w. Connor + def all_block_ids(self, cache_group: int = 0) -> torch.Tensor: + """ + Return the Tensor containing all block IDs for this sequence in the specified cache group. + + Arguments: + cache_group (int): The cache group to query. + """ + block_ids = [] + for allocation_group, num_blocks in zip(self._kv_cache_ids[cache_group], + self._blocks_per_allocation_group[cache_group]): + block_ids.append(allocation_group[:num_blocks]) + return torch.cat(block_ids) + + def pre_forward(self, num_tokens: int) -> None: + """ + Update the state of the sequence before a forward pass. + + Arguments: + num_tokens (int): The number of tokens in the sequence that will be executed during the + next forward pass of the model. + """ + self._in_flight_tokens = num_tokens + + def post_forward(self) -> None: + """ + Update the state of the sequence after a forward pass. This should be called after the forward + pass completes. NOTE: due to the asynchronous nature of the accelerator, this may be called + before the forward pass completes on the device itself. + """ + self._seen_tokens += self._in_flight_tokens + self._in_flight_tokens = 0 + + def extend_kv_cache(self, new_ids: Union[List[torch.IntTensor], torch.IntTensor], cache_group: int = 0) -> None: + """ + Extend the KV-cache for the sequence. + + Arguments: + new_ids (Union[List[torch.IntTensor], torch.IntTensor]): For each allocation group, the IDs + to add to the KV-cache. If there is only one allocation group, a single tensor can be + provided. Otherwise, a list of tensors should be provided. The tensors do not need + to have the same shape. + """ + if isinstance(new_ids, torch.Tensor): + new_ids = [new_ids] + + if len(new_ids) != self._num_allocation_groups[cache_group]: + raise ValueError( + f"Only {len(new_ids)} allocation groups provided, expected {self._num_allocation_groups[cache_group]}") + + for group_id, new_group_ids in enumerate(new_ids): + new_blocks = new_group_ids.numel() + + if new_blocks == 0: + # If we have multiple groups, it's possible to have an empty group. + continue + + shadow_alloc_group = self._kv_cache_ids_shadow[cache_group][group_id] + alloc_group = self._kv_cache_ids[cache_group][group_id] + cur_blocks = self._blocks_per_allocation_group[cache_group][group_id] + + shadow_alloc_group[cur_blocks:cur_blocks + new_blocks].copy_(new_group_ids) + alloc_group[cur_blocks:cur_blocks + new_blocks].copy_(shadow_alloc_group[cur_blocks:cur_blocks + + new_blocks], + non_blocking=True) + + self._blocks_per_allocation_group[cache_group][group_id] += new_blocks + + def free_kv_cache(self, free_ids: Union[List[torch.IntTensor], torch.IntTensor], cache_group: int = 0) -> None: + """ + Free blocks from the KV-cache for the sequence. + + Arguments: + free_ids (Union[List[torch.IntTensor], torch.IntTensor]): The ids of blocks to free + from the KV-cache. If there is only one allocation group, a single tensor can be + provided. Otherwise, a list of tensors should be provided. The tensors do not need + to have the same shape. + """ + raise NotImplementedError("Partial KV-cache freeing is not yet supported.") diff --git a/lib/python3.12/site-packages/deepspeed/inference/v2/scheduling_utils.py b/lib/python3.12/site-packages/deepspeed/inference/v2/scheduling_utils.py new file mode 100644 index 0000000000000000000000000000000000000000..6d3818d46675d9b00735454b3db021e770bb4947 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/inference/v2/scheduling_utils.py @@ -0,0 +1,54 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from enum import Enum + + +class SchedulingResult(Enum): + + Success = 0 + """ + The proposed batch is valid and can be scheduled. + """ + + EngineSequenceLimitExceeded = 1 + """ + The proposed batch would would overflow the number of concurrent sequences the engine may support. + """ + + BatchSequenceLimitExceeded = 2 + """ + The proposed batch contains more sequences than the engine was configured + to support in a single forwardp + """ + + BatchTokenLimitExceeded = 3 + """ + The proposed batch contains more tokens than the engine was configured + to support in a single forward. + """ + + KVCacheLimitExceeded = 4 + """ + The proposed batch would require more KV cache to be allocated than the engine + currently has available. + """ + + SequenceTokenLimitExceeded = 5 + """ + The proposed batch contains a sequence that is longer than the engine/model can support. + """ + + +class SchedulingError(RuntimeError): + + result: SchedulingResult + """ + The failed result of the scheduling check. Guaranteed to not be SchedulingResult.Success. + """ + + def __init__(self, result: SchedulingResult) -> None: + self.result = result + super().__init__(f"Batch scheduling failed with result {result}") diff --git a/lib/python3.12/site-packages/deepspeed/moe/__init__.py b/lib/python3.12/site-packages/deepspeed/moe/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..6c5067f71c8faf166bc78e88f9b62e8627dda7c7 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/moe/__init__.py @@ -0,0 +1,5 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team +'''Copyright The Microsoft DeepSpeed Team''' diff --git a/lib/python3.12/site-packages/deepspeed/moe/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2d5670a2bbf8225e3139e79ef1bd598a7e48c2d7 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/moe/__pycache__/experts.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/experts.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a9ad669de46633aa9e4aa2b6e4ca518deb144aa6 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/experts.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/moe/__pycache__/layer.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/layer.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d2374fd63636badca9505090afeeba1905691616 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/layer.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/moe/__pycache__/mappings.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/mappings.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..287c3dd3270b45d3bb4b219978cdff33b28e1d19 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/mappings.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/moe/__pycache__/sharded_moe.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/sharded_moe.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ef90066461084594e42625c4ec441c738997ef0b Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/sharded_moe.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/moe/__pycache__/utils.cpython-312.pyc b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/utils.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2875144f5d1d518134e80014bd5e00fb3a051452 Binary files /dev/null and b/lib/python3.12/site-packages/deepspeed/moe/__pycache__/utils.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/deepspeed/moe/experts.py b/lib/python3.12/site-packages/deepspeed/moe/experts.py new file mode 100644 index 0000000000000000000000000000000000000000..0863221d7edf93770af126e3fa1a78be3c7f4289 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/moe/experts.py @@ -0,0 +1,38 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +import copy +from typing import List, Optional + +import torch +from torch import nn + + +class Experts(nn.Module): + + def __init__(self, expert: nn.Module, num_local_experts: int = 1, expert_group_name: Optional[str] = None) -> None: + super(Experts, self).__init__() + + self.deepspeed_experts = nn.ModuleList([copy.deepcopy(expert) for _ in range(num_local_experts)]) + self.num_local_experts = num_local_experts + + # TODO: revisit allreduce for moe.gate... + for expert in self.deepspeed_experts: + # TODO: Create param groups to handle expert + data case (e.g. param.group = moe_group) + for param in expert.parameters(): + param.allreduce = False + param.group_name = expert_group_name + + def forward(self, inputs: torch.Tensor) -> torch.Tensor: + chunks = inputs.chunk(self.num_local_experts, dim=1) + expert_outputs: List[torch.Tensor] = [] + + for chunk, expert in zip(chunks, self.deepspeed_experts): + out = expert(chunk) + if isinstance(out, tuple): + out = out[0] # Ignore the bias term for now + expert_outputs += [out] + + return torch.cat(expert_outputs, dim=1) diff --git a/lib/python3.12/site-packages/deepspeed/moe/layer.py b/lib/python3.12/site-packages/deepspeed/moe/layer.py new file mode 100644 index 0000000000000000000000000000000000000000..6777788ab885a618544ced9ed9448e08c5159751 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/moe/layer.py @@ -0,0 +1,132 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from typing import Optional, Tuple + +import torch +from torch import nn +from torch.nn import functional as F + +from deepspeed.utils import groups, log_dist +from .experts import Experts +from .sharded_moe import MOELayer, TopKGate + + +class MoE(nn.Module): + """Initialize an MoE layer. + + Arguments: + hidden_size (int): the hidden dimension of the model, importantly this is also the input and output dimension. + expert (nn.Module): the torch module that defines the expert (e.g., MLP, torch.linear). + num_experts (int, optional): default=1, the total number of experts per layer. + ep_size (int, optional): default=1, number of ranks in the expert parallel world or group. + k (int, optional): default=1, top-k gating value, only supports k=1 or k=2. + capacity_factor (float, optional): default=1.0, the capacity of the expert at training time. + eval_capacity_factor (float, optional): default=1.0, the capacity of the expert at eval time. + min_capacity (int, optional): default=4, the minimum capacity per expert regardless of the capacity_factor. + use_residual (bool, optional): default=False, make this MoE layer a Residual MoE (https://arxiv.org/abs/2201.05596) layer. + noisy_gate_policy (str, optional): default=None, noisy gate policy, valid options are 'Jitter', 'RSample' or 'None'. + drop_tokens (bool, optional): default=True, whether to drop tokens - (setting to False is equivalent to infinite capacity). + use_rts (bool, optional): default=True, whether to use Random Token Selection. + use_tutel (bool, optional): default=False, whether to use Tutel optimizations (if installed). + enable_expert_tensor_parallelism (bool, optional): default=False, whether to use tensor parallelism for experts + top2_2nd_expert_sampling (bool, optional): default=True, whether to perform sampling for 2nd expert + """ + + def __init__(self, + hidden_size: int, + expert: nn.Module, + num_experts: int = 1, + ep_size: int = 1, + k: int = 1, + capacity_factor: float = 1.0, + eval_capacity_factor: float = 1.0, + min_capacity: int = 4, + use_residual: bool = False, + noisy_gate_policy: Optional[str] = None, + drop_tokens: bool = True, + use_rts: bool = True, + use_tutel: bool = False, + enable_expert_tensor_parallelism: bool = False, + top2_2nd_expert_sampling: bool = True) -> None: + + super(MoE, self).__init__() + + self.use_residual = use_residual + self.enable_expert_tensor_parallelism = enable_expert_tensor_parallelism + assert num_experts % ep_size == 0, f"Number of experts ({num_experts}) should be divisible by expert parallel size ({ep_size})" + self.ep_size = ep_size + self.expert_group_name = f"ep_size_{self.ep_size}" + self.num_experts = num_experts + self.num_local_experts = num_experts // self.ep_size + + log_dist( + f'Creating MoE layer with num_experts: {num_experts} | num_local_experts: {self.num_local_experts} | expert_parallel_size: {self.ep_size}', + [0]) + + assert noisy_gate_policy is None or noisy_gate_policy in ['None', 'Jitter', 'RSample'], \ + 'Unsupported noisy_gate_policy: ' + noisy_gate_policy + + experts = Experts(expert, self.num_local_experts, self.expert_group_name) + self.deepspeed_moe = MOELayer(TopKGate(hidden_size, num_experts, k, capacity_factor, eval_capacity_factor, + min_capacity, noisy_gate_policy, drop_tokens, use_rts, None, + top2_2nd_expert_sampling), + experts, + self.expert_group_name, + self.ep_size, + self.num_local_experts, + use_tutel=use_tutel) + if self.use_residual: + self.mlp = expert + # coefficient is used for weighted sum of the output of expert and mlp + self.coefficient = nn.Linear(hidden_size, 2) + + def set_deepspeed_parallelism(self, use_data_before_expert_parallel_: bool = False) -> None: + self._create_process_groups(use_data_before_expert_parallel_=use_data_before_expert_parallel_) + + def _create_process_groups(self, use_data_before_expert_parallel_: bool = False) -> None: + # Create process group for a layer if needed + if self.expert_group_name not in groups._get_expert_parallel_group_dict(): + print(f"No existing process group found, creating a new group named: {self.expert_group_name}") + if (groups.mpu is None) or (not self.enable_expert_tensor_parallelism): + # Condition 1 - no groups.mpu means no tensor parallelism + # Condition 2 - disabling expert tensor parallelism on purpose + groups._create_expert_and_data_parallel( + self.ep_size, use_data_before_expert_parallel_=use_data_before_expert_parallel_) + else: + # expert tensor parallelism is enabled + groups._create_expert_data_and_model_parallel( + self.ep_size, mpu=groups.mpu, use_data_before_expert_parallel_=use_data_before_expert_parallel_) + # Set the group handle for the MOELayer (deepspeed_moe) object + self.deepspeed_moe._set_ep_group(groups._get_expert_parallel_group(self.expert_group_name)) + + def forward(self, + hidden_states: torch.Tensor, + used_token: Optional[torch.Tensor] = None) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor]: + """ MoE forward + + Arguments: + hidden_states (Tensor): input to the layer + used_token (Tensor, optional): default: None, mask only used tokens + + Returns: + A tuple including output, gate loss, and expert count. + + * output (Tensor): output of the model + + * l_aux (Tensor): gate loss value + + * exp_counts (Tensor): expert count + """ + output = self.deepspeed_moe(hidden_states, used_token) + if self.use_residual: + # Residual MoE + output_mlp = self.mlp(hidden_states) + if isinstance(output_mlp, tuple): + output_mlp = output_mlp[0] # Ignore the bias term for now + coef = self.coefficient(hidden_states) + coef = F.softmax(coef, dim=-1) + output = output * coef[..., 0:1] + output_mlp * coef[..., 1:] + return output, self.deepspeed_moe.l_aux, self.deepspeed_moe.exp_counts diff --git a/lib/python3.12/site-packages/deepspeed/moe/mappings.py b/lib/python3.12/site-packages/deepspeed/moe/mappings.py new file mode 100644 index 0000000000000000000000000000000000000000..e57f66b85193d86734c186a916baa6da033a90a1 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/moe/mappings.py @@ -0,0 +1,118 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# The file has been adapted from the following Megatron-LM file: +# https://github.com/NVIDIA/Megatron-LM/blob/main/megatron/mpu/mappings.py +# Git commit hash: 9dc3c42a84aa656f583703cf8b6b4f79f712b796 +# We retain the following copyright from the original files: + +# Copyright (c) 2020, NVIDIA CORPORATION. All rights reserved. +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import torch +import deepspeed +from deepspeed.utils.bwc import (bwc_tensor_model_parallel_world_size, bwc_tensor_model_parallel_rank, + bwc_tensor_model_parallel_group) + + +def _gather_tokens(input_, dim=0): + """Gather tensors and concatenate them along a dimension""" + mpu = deepspeed.utils.groups.mpu + + input_ = input_.contiguous() + world_size = bwc_tensor_model_parallel_world_size(mpu) + if world_size == 1: + return input_ + + gather_buffer = torch.empty(world_size * input_.numel(), dtype=input_.dtype, device=input_.device) + deepspeed.comm.all_gather_into_tensor(gather_buffer, input_, group=bwc_tensor_model_parallel_group(mpu)) + if dim == 0: + shape = list(input_.size()) + shape[0] = shape[0] * world_size + output = gather_buffer.view(shape) + else: + tensor_list = [ + gather_buffer.narrow(0, + input_.numel() * i, input_.numel()).view_as(input_) for i in range(world_size) + ] + # Note: torch.cat already creates a contiguous tensor. + output = torch.cat(tensor_list, dim=dim).contiguous() + + return output + + +def _drop_tokens(input_, dim=0): + """Divide a tensor among the tensor parallel ranks""" + mpu = deepspeed.utils.groups.mpu + + total_chunks = bwc_tensor_model_parallel_world_size(mpu) + if total_chunks == 1: + return input_ + this_chunk = bwc_tensor_model_parallel_rank(mpu) + assert input_.shape[ + dim] % total_chunks == 0, f"input dimension {dim} ({input_.shape[dim]}) is not divisible by tensor parallel world size ({total_chunks})" + chunk_size = input_.shape[dim] // total_chunks + + return torch.narrow(input_, dim, this_chunk * chunk_size, chunk_size) + + +class _GatherTokens(torch.autograd.Function): + """All gather tokens among the tensor parallel ranks""" + + @staticmethod + def symbolic(graph, input_, dim): + return _gather_tokens(input_, dim) + + @staticmethod + def forward(ctx, input_, dim): + ctx.dim = dim + return _gather_tokens(input_, dim) + + @staticmethod + def backward(ctx, grad_output): + return _drop_tokens(grad_output, ctx.dim), None + + +class _DropTokens(torch.autograd.Function): + "Divide tokens equally among the tensor parallel ranks" + + @staticmethod + def symbolic(graph, input_, dim): + return _drop_tokens(input_, dim) + + @staticmethod + def forward(ctx, input_, dim): + ctx.dim = dim + return _drop_tokens(input_, dim) + + @staticmethod + def backward(ctx, input_): + return _gather_tokens(input_, ctx.dim), None + + +def gather_tokens(input_, dim=0): + mpu = deepspeed.utils.groups.mpu + if mpu is None or bwc_tensor_model_parallel_world_size(mpu) == 1: + # no tensor parallelism for non-experts + return input_ + return _GatherTokens.apply(input_, dim) + + +def drop_tokens(input_, dim=0): + mpu = deepspeed.utils.groups.mpu + if mpu is None or bwc_tensor_model_parallel_world_size(mpu) == 1: + # no tensor parallelism for non-experts + return input_ + return _DropTokens.apply(input_, dim) diff --git a/lib/python3.12/site-packages/deepspeed/moe/sharded_moe.py b/lib/python3.12/site-packages/deepspeed/moe/sharded_moe.py new file mode 100644 index 0000000000000000000000000000000000000000..a00d694fbc1485a0d35b2938e489c2b1a9f5e561 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/moe/sharded_moe.py @@ -0,0 +1,677 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team +""" +The file has been adapted from two fairscale files: + (1) https://github.com/facebookresearch/fairscale/blob/master/fairscale/nn/moe/moe_layer.py + (2) https://github.com/facebookresearch/fairscale/blob/master/fairscale/nn/moe/top2gate.py + Git commit hash: 34df606902a240567a0d898037ece55c2f1336cf + We retain the following license from the original files: +""" + +# Copyright (c) Facebook, Inc. and its affiliates. All rights reserved. +# +# This source code is licensed under the BSD license found in the +# LICENSE file in the root directory of this source tree. + +from deepspeed.utils.timer import SynchronizedWallClockTimer +from deepspeed.utils import logger +from deepspeed.utils.bwc import bwc_tensor_model_parallel_world_size +from typing import Callable, Dict, TYPE_CHECKING, Any, Optional, Tuple, Union + +import torch +from torch import Tensor +from torch.nn import Module +import torch.nn.functional as F +from deepspeed.utils import groups +from .mappings import drop_tokens, gather_tokens + +if TYPE_CHECKING: + Base = Module[Tensor] +else: + Base = Module + +TOPK_GATE_TIMER = 'topk_gate' +MOE_TIMER = 'moe' +FIRST_ALLTOALL_TIMER = '1st_a2a' +SECOND_ALLTOALL_TIMER = '2nd_a2a' + +uniform_map: Dict[torch.device, Callable] = {} +gumbel_map: Dict[torch.device, Callable] = {} +exp_selection_uniform_map: Dict[torch.device, Callable] = {} + +try: + # To enable Tutel MoE optimizations: + # python3 -m pip install --user --upgrade git+https://github.com/deepspeedai/tutel@v0.1.x + from tutel import moe as tutel_moe + TUTEL_INSTALLED = True +except: + # Fail silently so we don't spam logs unnecessarily if user isn't using tutel + TUTEL_INSTALLED = False + pass + + +def multiplicative_jitter(x, device: torch.device, epsilon=1e-2): + """ + Modified from switch transformer paper. mesh transformers + Multiply values by a random number between 1-epsilon and 1+epsilon. + Makes models more resilient to rounding errors introduced by bfloat16. + This seems particularly important for logits. + Args: + x: a torch.tensor + device: torch.device + epsilon: a floating point value + Returns: + a jittered x. + """ + if epsilon == 0: + return x + uniform = uniform_map.get(device) + if uniform is None: + uniform = torch.distributions.uniform.Uniform(low=torch.tensor(1.0 - epsilon, device=device), + high=torch.tensor(1.0 + epsilon, + device=device)).rsample # type: ignore + uniform_map[device] = uniform + return x * uniform(x.shape) + + +def gumbel_rsample(shape: Tuple, device: torch.device) -> Tensor: + gumbel = gumbel_map.get(device) + if gumbel is None: + one = torch.tensor(1.0, device=device) + zero = torch.tensor(0.0, device=device) + gumbel = torch.distributions.gumbel.Gumbel(zero, one).rsample # type: ignore + gumbel_map[device] = gumbel + return gumbel(shape) + + +from deepspeed import comm as dist + +# einsum dimensions: (g)roup, (s)equence, (e)xpert, (m)odel, (c)apacity +# See https://arxiv.org/pdf/2006.16668.pdf for details. + + +# Based on https://github.com/pytorch/pytorch/pull/40762 +class _AllToAll(torch.autograd.Function): + + @staticmethod + def forward(ctx: Any, group: dist.ProcessGroup, input: Tensor) -> Tensor: # type: ignore + ctx.group = group + input = input.contiguous() + output = torch.empty_like(input) + dist.all_to_all_single(output, input, group=group) + return output + + @staticmethod + def backward(ctx: Any, *grad_output: Tensor) -> Tuple[None, Tensor]: + return (None, _AllToAll.apply(ctx.group, *grad_output)) + + +# einsum rewrites are on par or more performant +# switch can be bubbled up in future +USE_EINSUM = True + + +# einsum dimensions: (g)roup, (s)equence, (e)xpert, (m)odel, (c)apacity +# See https://arxiv.org/pdf/2006.16668.pdf for details. +def einsum(rule, a, b): + if USE_EINSUM: + return torch.einsum(rule, a, b) + elif rule == 's,se->se': + return a.reshape(a.shape[0], -1) * b + elif rule == 'se,sc->sec': + return a.unsqueeze(2) * b.unsqueeze(1) + elif rule == 'se,se->s': + return torch.bmm(a.unsqueeze(1), b.unsqueeze(2)).reshape(-1) + elif rule == 'se,sec->sec': + return a.unsqueeze(2) * b + elif rule == 'sec,sm->ecm': + s = a.shape[0] + e = a.shape[1] + c = a.shape[2] + m = b.shape[1] + return torch.matmul(a.reshape(s, -1).t(), b).reshape(e, c, m) + elif rule == 'sec,ecm->sm': + return torch.matmul(a.reshape(a.shape[0], -1), b.reshape(-1, b.shape[-1])) + elif rule == 'ks,ksm->sm': + k = b.shape[0] + s = b.shape[1] + m = b.shape[2] + # [k, s] -> [s, k] -> [s, 1, k] + a = a.t().unsqueeze(1) + # [k,s,m] -> [k, sm] -> [sm, k] -> [s, m, k] + b = b.reshape(k, -1).t().reshape(s, m, k) + # bmm([s, 1, k], [s, m, k]^t) -> [s, m, 1] + return torch.bmm(a, b.transpose(1, 2)).squeeze(2) + else: + return torch.einsum(rule, a, b) + + +# The following functions are extracted and scripted +# because otherwise during a torch.jit.trace, the non-Tensor +# values used in the calculations get recorded as constants. +# torch.jit.script coerces them into Tensors and preserves +# their dynamic shapes. This enables ONNX export. +# We can't script the entire top1gating function because it +# includes stateful caching logic which is incompatible with ONNX. + + +@torch.jit.script +def _capacity(gates: Tensor, capacity_factor: Tensor, min_capacity: Tensor) -> Tensor: + # gates has shape of SE + num_tokens = gates.shape[0] + num_experts = gates.shape[1] + # to(torch.int64) works around a bug in torch.onnx.export: + # it should cast k to int64 when converting torch.topk but it doesn't. + capacity = torch.ceil((num_tokens / num_experts) * capacity_factor).to(torch.int64) + if capacity < min_capacity: + capacity = min_capacity.to(torch.int64) + return capacity + + +@torch.jit.script +def _top_idx(source, k): + return torch.topk(source, k=k, dim=0)[1] + + +@torch.jit.script +def _one_hot_to_float(x, num_classes): + return F.one_hot(x, num_classes=num_classes).float() + + +def top1gating(logits: Tensor, + capacity_factor: float, + min_capacity: int, + used_token: Tensor = None, + noisy_gate_policy: Optional[str] = None, + drop_tokens: bool = True, + use_rts: bool = True, + ep_group: Union[torch.distributed.ProcessGroup, None] = None, + use_tutel: bool = False) -> Tuple[Tensor, Tensor, Tensor, Tensor]: + """Implements Top1Gating on logits.""" + if noisy_gate_policy == 'RSample': + logits_w_noise = logits + gumbel_rsample(logits.shape, device=logits.device) + # everything is in fp32 in this function + + gates = F.softmax(logits, dim=1) + capacity = _capacity(gates, torch.tensor(capacity_factor), torch.tensor(min_capacity)) + + # Create a mask for 1st's expert per token + # noisy gating + indices1_s = torch.argmax(logits_w_noise if noisy_gate_policy == 'RSample' else gates, dim=1) + num_experts = int(gates.shape[1]) + mask1 = F.one_hot(indices1_s, num_classes=num_experts) + + # mask only used tokens + if used_token is not None: + mask1 = einsum("s,se->se", used_token, mask1) + + # gating decisions + exp_counts = torch.sum(mask1, dim=0).detach().to(logits.device) + + # if we don't want to drop any tokens + if not drop_tokens: + new_capacity = torch.max(exp_counts).to(logits.device) + # Communicate across expert processes to pick the maximum capacity. + if ep_group is not None: + dist.all_reduce(new_capacity, op=dist.ReduceOp.MAX, group=ep_group) + if groups._get_expert_model_parallel_world_size() == 1: + # If the non-expert is tensor-parallel, we need to pad the capacity to 'tp'. + # This is since we are going to activate drop_tokens() to drop duplicate tokens. + tp = 1 if groups.mpu is None else bwc_tensor_model_parallel_world_size(mpu=groups.mpu) + new_capacity = torch.ceil(new_capacity / tp).mul(tp).to(new_capacity.dtype) + # Make sure the capacity value does not exceed the number of tokens. + capacity = min(new_capacity, torch.tensor(mask1.size(0)).to(new_capacity.device)) + + # Compute l_aux + me = torch.mean(gates, dim=0) + ce = torch.mean(mask1.float(), dim=0) + l_aux = torch.sum(me * ce) * num_experts + + # Random Token Selection + if use_rts: + uniform = exp_selection_uniform_map.get(logits.device) + if uniform is None: + uniform = torch.distributions.uniform.Uniform(low=torch.tensor(0.0, device=logits.device), + high=torch.tensor(1.0, device=logits.device)).rsample + exp_selection_uniform_map[logits.device] = uniform + + mask1_rand = mask1 * uniform(mask1.shape) + else: + mask1_rand = mask1 + + assert logits.shape[ + 0] >= min_capacity, "No. of tokens (batch-size) should be greater than min_capacity. Either set min_capacity to 0 or increase your batch size." + + top_idx = _top_idx(mask1_rand, capacity) + + new_mask1 = mask1 * torch.zeros_like(mask1).scatter_(0, top_idx, 1) + mask1 = new_mask1 + + if use_tutel: + # Tutel doesn't support index values masked with zero + # so we need to replace masked indices with -1 + indices_mask = mask1.sum(dim=1) * num_experts - 1 + indices1_s = torch.min(indices1_s, indices_mask) + + # Compute locations in capacity buffer + if use_tutel: + locations1 = tutel_moe.fast_cumsum_sub_one(mask1) + else: + locations1 = torch.cumsum(mask1, dim=0) - 1 + + if use_tutel: + gates1_s = (gates * mask1).sum(dim=1) + locations1_s = torch.sum(locations1 * mask1, dim=1) + return l_aux, capacity, num_experts, [ + indices1_s, + ], [ + locations1_s, + ], [ + gates1_s, + ], exp_counts + + # Store the capacity location for each token + locations1_s = torch.sum(locations1 * mask1, dim=1) + + # Normalize gate probabilities + mask1_float = mask1.float() + gates = gates * mask1_float + + locations1_sc = _one_hot_to_float(locations1_s, capacity) + combine_weights = einsum("se,sc->sec", gates, locations1_sc) + + dispatch_mask = combine_weights.bool() + + return l_aux, combine_weights, dispatch_mask, exp_counts + + +def top2gating(logits: Tensor, + capacity_factor: float, + min_capacity: int, + drop_tokens: bool = True, + ep_group: Union[torch.distributed.ProcessGroup, None] = None, + top2_2nd_expert_sampling: bool = True) -> Tuple[Tensor, Tensor, Tensor, Tensor]: + """Implements Top2Gating on logits.""" + # everything is in fp32 in this function + gates = F.softmax(logits, dim=1) + + # Create a mask for 1st's expert per token + indices1_s = torch.argmax(gates, dim=1) + num_experts = int(gates.shape[1]) + mask1 = F.one_hot(indices1_s, num_classes=num_experts) + + if top2_2nd_expert_sampling: + # Create a mask for 2nd's expert per token using Gumbel-max trick + # https://timvieira.github.io/blog/post/2014/07/31/gumbel-max-trick/ + logits += gumbel_rsample(logits.shape, device=logits.device) + + # Replace top-expert with min value + logits_except1 = logits.masked_fill(mask1.bool(), float("-inf")) + indices2_s = torch.argmax(logits_except1, dim=1) + mask2 = F.one_hot(indices2_s, num_classes=num_experts) + + # Compute locations in capacity buffer + locations1 = torch.cumsum(mask1, dim=0) - 1 + locations2 = torch.cumsum(mask2, dim=0) - 1 + # Update 2nd's location by accounting for locations of 1st + locations2 += torch.sum(mask1, dim=0, keepdim=True) + + # Compute l_aux + me = torch.mean(gates, dim=0) + ce = torch.mean(mask1.float(), dim=0) + l_aux = torch.mean(me * ce) * num_experts * num_experts + + # gating decisions + exp_counts = torch.sum(mask1 + mask2, dim=0).detach().to(logits.device) + + if drop_tokens: + # Calculate configured capacity and remove locations outside capacity from mask + capacity = _capacity(gates, torch.tensor(capacity_factor * 2), torch.tensor(min_capacity)) + mask1 *= torch.lt(locations1, capacity) + mask2 *= torch.lt(locations2, capacity) + else: + # Do not drop tokens - set capacity according to current expert assignments + new_capacity = torch.max(exp_counts) + if ep_group is not None: + dist.all_reduce(new_capacity, op=dist.ReduceOp.MAX, group=ep_group) + if groups._get_expert_model_parallel_world_size() == 1: + # If the non-expert is tensor-parallel, we need to pad the capacity to 'tp'. + # This is since we are going to activate drop_tokens() to drop duplicate tokens. + tp = 1 if groups.mpu is None else bwc_tensor_model_parallel_world_size(mpu=groups.mpu) + new_capacity = torch.ceil(new_capacity / tp).mul(tp).to(new_capacity.dtype) + capacity = new_capacity + + # Store the capacity location for each token + locations1_s = torch.sum(locations1 * mask1, dim=1) + locations2_s = torch.sum(locations2 * mask2, dim=1) + + # Normalize gate probabilities + mask1_float = mask1.float() + mask2_float = mask2.float() + gates1_s = einsum("se,se->s", gates, mask1_float) + gates2_s = einsum("se,se->s", gates, mask2_float) + denom_s = gates1_s + gates2_s + # Avoid divide-by-zero + denom_s = torch.clamp(denom_s, min=torch.finfo(denom_s.dtype).eps) + gates1_s /= denom_s + gates2_s /= denom_s + + # Calculate combine_weights and dispatch_mask + gates1 = einsum("s,se->se", gates1_s, mask1_float) + gates2 = einsum("s,se->se", gates2_s, mask2_float) + locations1_sc = _one_hot_to_float(locations1_s, capacity) + locations2_sc = _one_hot_to_float(locations2_s, capacity) + combine1_sec = einsum("se,sc->sec", gates1, locations1_sc) + combine2_sec = einsum("se,sc->sec", gates2, locations2_sc) + combine_weights = combine1_sec + combine2_sec + dispatch_mask = combine_weights.bool() + + return l_aux, combine_weights, dispatch_mask, exp_counts + + +def topkgating( + logits: Tensor, + k: int, + capacity_factor: float, + min_capacity: int, + drop_tokens: bool = True, + ep_group: Union[torch.distributed.ProcessGroup, None] = None, + drop_policy: str = "probs", +) -> Tuple[Tensor, Tensor, Tensor, Tensor]: + """Implements TopKGating on logits.""" + + # everything is in fp32 in this function + # get topk gates + top_gate, top_idx = torch.topk(logits, k=k, dim=1) + # gating decisions + gates = F.softmax(logits, dim=1) + num_experts = int(gates.shape[1]) + + # get topk mask + topk_masked_gates = torch.zeros_like(logits).scatter(1, top_idx, top_gate) + + mask = torch.zeros_like(gates, dtype=torch.bool).scatter_(1, top_idx, 1) + + exp_counts = torch.sum(mask, dim=0).detach().to(logits.device) + + # Compute l_aux + me = torch.mean(gates, dim=0) + ce = torch.mean(mask.float(), dim=0) + l_aux = torch.mean(me * ce) * num_experts * num_experts / k + + if drop_tokens: + # Calculate configured capacity and remove locations outside capacity from mask + capacity = _capacity(gates, torch.tensor(capacity_factor * k), torch.tensor(min_capacity)) + # update mask and locations by capacity + + if drop_policy == 'probs': + capacity_probs, capacity_indices = torch.topk(topk_masked_gates, k=capacity, dim=0, sorted=False) + capacity_mask = torch.zeros_like(logits).scatter(0, capacity_indices, 1) + mask = torch.logical_and(mask, capacity_mask) + locations = torch.cumsum(mask, dim=0) - 1 + + elif drop_policy == "position": + locations = torch.cumsum(mask, dim=0) - 1 + mask *= torch.lt(locations, capacity) + else: + raise ValueError(f"Invalid drop_policy: {drop_policy}") + + else: + # Do not drop tokens - set capacity according to current expert assignments + new_capacity = torch.max(exp_counts) + if ep_group is not None: + dist.all_reduce(new_capacity, op=dist.ReduceOp.MAX, group=ep_group) + if groups._get_expert_model_parallel_world_size() == 1: + # If the non-expert is tensor-parallel, we need to pad the capacity to 'tp'. + # This is since we are going to activate drop_tokens() to drop duplicate tokens. + tp = 1 if groups.mpu is None else bwc_tensor_model_parallel_world_size(mpu=groups.mpu) + new_capacity = torch.ceil(new_capacity / tp).mul(tp).to(new_capacity.dtype) + capacity = new_capacity + + # normalize gates + gates_masked = gates * mask + gates_s = torch.sum(gates_masked, dim=-1, keepdim=True) + denom_s = torch.clamp(gates_s, min=torch.finfo(gates_masked.dtype).eps) + gates_masked = gates_masked / denom_s + + # dispatch_mask + locations_sc = _one_hot_to_float((locations * mask), capacity) + + combine_weights = torch.einsum("se,sec->sec", gates_masked, locations_sc) + + dispatch_mask = combine_weights.bool() + + return l_aux, combine_weights, dispatch_mask, exp_counts + + +class TopKGate(Module): + """Gate module which implements Top2Gating as described in Gshard_. + :: + + gate = TopKGate(model_dim, num_experts) + l_aux, combine_weights, dispatch_mask = gate(input) + + .. Gshard_: https://arxiv.org/pdf/2006.16668.pdf + + Args: + model_dim (int): + size of model embedding dimension + num_experts (int): + number of experts in model + """ + + wg: torch.nn.Linear + + def __init__(self, + model_dim: int, + num_experts: int, + k: int = 1, + capacity_factor: float = 1.0, + eval_capacity_factor: float = 1.0, + min_capacity: int = 8, + noisy_gate_policy: Optional[str] = None, + drop_tokens: bool = True, + use_rts: bool = True, + ep_group: Union[torch.distributed.ProcessGroup, None] = None, + top2_2nd_expert_sampling: bool = True) -> None: + super().__init__() + + self.wg = torch.nn.Linear(model_dim, num_experts, bias=False) + self.ep_group = ep_group + self.k = k + self.capacity_factor = capacity_factor + self.eval_capacity_factor = eval_capacity_factor + self.min_capacity = min_capacity + self.noisy_gate_policy = noisy_gate_policy + self.timers = SynchronizedWallClockTimer() + self.wall_clock_breakdown = False + self.gate_time = 0.0 + self.drop_tokens = drop_tokens + self.use_rts = use_rts + self.top2_2nd_expert_sampling = top2_2nd_expert_sampling + + def _set_ep_group(self, ep_group): + assert self.ep_group is None, f'Attempting to override an existing ep_group' + self.ep_group = ep_group + + def forward(self, + input: torch.Tensor, + used_token: torch.Tensor = None, + use_tutel: bool = False) -> Tuple[Tensor, Tensor, Tensor]: # type: ignore + + if self.wall_clock_breakdown: + self.timers(TOPK_GATE_TIMER).start() + + input_fp32 = input.float() + # input jittering + if self.noisy_gate_policy == 'Jitter' and self.training: + input_fp32 = multiplicative_jitter(input_fp32, device=input.device) + logits = torch.nn.functional.linear(input_fp32, weight=self.wg.weight.float(), bias=None) + + if self.k == 1: + gate_output = top1gating(logits, self.capacity_factor if self.training else self.eval_capacity_factor, + self.min_capacity, used_token, self.noisy_gate_policy if self.training else None, + self.drop_tokens, self.use_rts, self.ep_group, use_tutel) + + elif self.k == 2: + gate_output = top2gating(logits, self.capacity_factor if self.training else self.eval_capacity_factor, + self.min_capacity, self.drop_tokens, self.ep_group, self.top2_2nd_expert_sampling) + else: + gate_output = topkgating(logits, self.k, + self.capacity_factor if self.training else self.eval_capacity_factor, + self.min_capacity, self.drop_tokens, self.ep_group) + + if self.wall_clock_breakdown: + self.timers(TOPK_GATE_TIMER).stop() + self.gate_time = self.timers(TOPK_GATE_TIMER).elapsed(reset=False) + + return gate_output + + +class MOELayer(Base): + """MOELayer module which implements MixtureOfExperts as described in Gshard_. + :: + + gate = TopKGate(model_dim, num_experts) + moe = MOELayer(gate, expert) + output = moe(input) + l_aux = moe.l_aux + + .. Gshard_: https://arxiv.org/pdf/2006.16668.pdf + + Args: + gate (torch.nn.Module): + gate network + expert (torch.nn.Module): + expert network + """ + + def __init__(self, + gate: Module, + experts: Module, + ep_group_name, + ep_size, + num_local_experts: int, + use_tutel: bool = False) -> None: + super().__init__() + self.gate = gate + self.experts = experts + self.ep_group = None + self.ep_size = ep_size + self.ep_group_name = ep_group_name + self.num_local_experts = num_local_experts + self.time_falltoall = 0.0 + self.time_salltoall = 0.0 + self.time_moe = 0.0 + self.timers = SynchronizedWallClockTimer() + self.wall_clock_breakdown = False + + self.use_tutel = use_tutel and TUTEL_INSTALLED and gate.k == 1 + + if self.use_tutel: + logger.info('Using Tutel optimizations.') + elif use_tutel and not TUTEL_INSTALLED: + logger.warning("Tutel optimization requested but not installed. " + "Proceeding without Tutel.") + elif use_tutel and TUTEL_INSTALLED and gate.k != 1: + logger.warning("To enable Tutel optimization, use top-1 instead of top-2 gate. " + "Proceeding without Tutel.") + + def _set_ep_group(self, ep_group): + self.ep_group = ep_group + self.gate._set_ep_group(ep_group) + + def forward(self, *input: Tensor, **kwargs: Any) -> Tensor: + + if self.wall_clock_breakdown: + self.timers(MOE_TIMER).start() + + # Implement Algorithm 2 from GShard paper. + d_model = input[0].shape[-1] + + # Initial implementation -> Reshape into S tokens by dropping sequence dimension. + # Reshape into G groups so that each group can distribute tokens equally + # group_size = kwargs['group_size'] if 'group_size' in kwargs.keys() else 1 + reshaped_input = input[0].reshape(-1, d_model) + + if self.use_tutel: + self.l_aux, C, E, indices_, locations_, gates_, self.exp_counts = self.gate(reshaped_input, input[1], True) + S, M = reshaped_input.size(0), reshaped_input.size(1) + + if not hasattr(self, '_tutel_dispatcher'): + self._tutel_dispatcher = tutel_moe.fast_dispatcher(E, C, M, dispatch_dtype=reshaped_input.dtype) + self._tutel_dispatcher.update(indices_, locations_, gates_, capacity=C) + dispatched_input = self._tutel_dispatcher.encode(reshaped_input) + else: + self.l_aux, combine_weights, dispatch_mask, self.exp_counts = self.gate(reshaped_input, input[1]) + dispatched_input = einsum("sec,sm->ecm", dispatch_mask.type_as(input[0]), reshaped_input) + + if self.wall_clock_breakdown: + self.timers(FIRST_ALLTOALL_TIMER).start() + + tensor_model_world_size = bwc_tensor_model_parallel_world_size(groups.mpu) + if tensor_model_world_size > 1: + # If the non-expert is tensor-parallel, + # Whether expert is tensor-parallel or not , it will create + # duplicate tokens on the tensor-parallel ranks. + # drop duplicate tokens also doubles up as a communication + # optimization as we are reducing the all-to-all communication volume. + # 1: for not tensor-parallel expert,drop duplicate tokens to ensure + # both correctness and reduce all-to-all communication. + # 2: for tensor-parallel expert,drop duplicate tokens to reduce all-to-all + # communication volume,before expert execution, it is necessary to perform + # an allgather to ensure correctness, + dispatched_input = drop_tokens(dispatched_input, dim=1) + + dispatched_input = _AllToAll.apply(self.ep_group, dispatched_input) + + if self.wall_clock_breakdown: + self.timers(FIRST_ALLTOALL_TIMER).stop() + self.time_falltoall = self.timers(FIRST_ALLTOALL_TIMER).elapsed(reset=False) + + if tensor_model_world_size > 1 and groups._get_expert_model_parallel_world_size() > 1: + # if both expert and non-expert are tensor-parallel + # the dropped duplicate tokens need to be gathered on each + # tensor parallel rank again to ensure correctness + dispatched_input = gather_tokens(dispatched_input, dim=1) + + # Re-shape after all-to-all: ecm -> gecm + dispatched_input = dispatched_input.reshape(self.ep_size, self.num_local_experts, -1, d_model) + expert_output = self.experts(dispatched_input) + # Re-shape before drop_tokens: gecm -> ecm + expert_output = expert_output.reshape(self.ep_size * self.num_local_experts, -1, d_model) + if tensor_model_world_size > 1 and groups._get_expert_model_parallel_world_size() > 1: + # if both expert and non-expert are tensor-parallel + # drop duplicate tokens to ensure both correctness + # and reduce all-to-all communication. + expert_output = drop_tokens(expert_output, dim=1) + + if self.wall_clock_breakdown: + self.timers(SECOND_ALLTOALL_TIMER).start() + + expert_output = _AllToAll.apply(self.ep_group, expert_output) + + if self.wall_clock_breakdown: + self.timers(SECOND_ALLTOALL_TIMER).stop() + self.time_salltoall = self.timers(SECOND_ALLTOALL_TIMER).elapsed(reset=False) + + if tensor_model_world_size > 1: + # the dropped duplicate tokens need to be gathered on each + # tensor parallel rank again for the tensor-parallel + # non-expert of the next layer. + expert_output = gather_tokens(expert_output, dim=1) + + if self.use_tutel: + combined_output = self._tutel_dispatcher.decode(expert_output.view(E * C, M)) + else: + combined_output = einsum("sec,ecm->sm", combine_weights.type_as(input[0]), expert_output) + + a = combined_output.reshape(input[0].shape) + + if self.wall_clock_breakdown: + self.timers(MOE_TIMER).stop() + self.time_moe = self.timers(MOE_TIMER).elapsed(reset=False) + + return a diff --git a/lib/python3.12/site-packages/deepspeed/moe/utils.py b/lib/python3.12/site-packages/deepspeed/moe/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..20866378efac72c96a7e2f56bdf97e5b7b4effb9 --- /dev/null +++ b/lib/python3.12/site-packages/deepspeed/moe/utils.py @@ -0,0 +1,182 @@ +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +from collections import defaultdict +from typing import Any, Dict, List, Set, Tuple, Union, cast + +import torch +from torch import nn + +from .layer import MoE + + +def has_moe_layers(m: nn.Module) -> Tuple[bool, int]: + has_moe = False + num_experts = 0 + + for module in m.modules(): + if isinstance(module, MoE): + has_moe = True + num_experts = module.num_experts + break + return has_moe, num_experts + + +def is_moe_param(param: torch.Tensor) -> bool: + if hasattr(param, "allreduce") and not param.allreduce: + return True + return False + + +def split_params_into_shared_and_expert_params( + params: List[torch.nn.Parameter]) -> Tuple[List[torch.nn.Parameter], List[torch.nn.Parameter]]: + shared_params: List[nn.Parameter] = [] + expert_params: List[nn.Parameter] = [] + + for p in params: + if is_moe_param(p): + expert_params.append(p) + else: + shared_params.append(p) + return shared_params, expert_params + + +def split_params_grads_into_shared_and_expert_params( + group: List[torch.nn.Parameter]) -> Tuple[List[torch.Tensor], List[torch.Tensor]]: + """Split grad of parameters into grads of non-expert params + and grads of expert params. This is useful while computing + grad-norms for clipping and overflow detection + + group (List[torch.nn.Parameter]): + Args: + The group of parameters to split + + Returns: + Tuple[List[torch.Tensor], List[torch.Tensor]]: + list of gradients for non MoE params, list of gradients of MoE params + """ + expert_grads: List[torch.Tensor] = [] + shared_grads: List[torch.Tensor] = [] + + for p in group: + if p.grad is not None: + if is_moe_param(p): + expert_grads.append(p.grad.to(p.dtype)) + else: + shared_grads.append(p.grad.to(p.dtype)) + return shared_grads, expert_grads + + +def split_params_into_different_moe_groups_for_optimizer( + param_groups: Union[Dict[str, Any], Tuple[Dict[str, Any], ...], List[Dict[str, Any]]], + max_group_size: Union[int, float] = 178956971) -> List[Dict[str, Any]]: + """Split parameters into different MoE groups for optimizer + + Args: + param_groups (Union[Dict[str, Any], Tuple[Dict[str, Any], ...], List[Dict[str, Any]]]) + The list of parameter groups to split + + Returns: + List[Dict[str, Any]]: + list of MoE/non-MoE groups for optimizer + """ + if isinstance(param_groups, tuple): + param_groups = list(param_groups) # Tuple cannot be modified + elif isinstance(param_groups, dict): + param_groups = [param_groups] + elif not isinstance(param_groups, list): + raise ValueError(f"Unknown param group type of {type(param_groups)}") + + # gather all data parallel group names + data_parallel_group_names: Set[str] = set() + for param_group in param_groups: + for param in cast(List[nn.Parameter], param_group["params"]): + if is_moe_param(param): + data_parallel_group_names.add(param.group_name) + + # Create the param MoE groups, leave param assign to next step + group_moe: Dict[str, Dict[str, Dict[str, Any]]] = defaultdict(lambda: defaultdict(dict)) + for param_group in param_groups: + for key in data_parallel_group_names: + group_moe[param_group['name']][key] = { + **param_group, + 'name': key, + 'moe': True, + 'params': [], + } + + # Assign param + for param_group in param_groups: + new_params: List[nn.Parameter] = [] + + for param in cast(List[nn.Parameter], param_group['params']): + if is_moe_param(param): + group_moe[param_group['name']][param.group_name]['params'].append(param) + else: + new_params.append(param) + param_group['params'] = new_params + + # Flatten the moe groups + if max_group_size is not None: + for moe_group in group_moe.values(): + for param_group in moe_group.values(): + cur_group: List[nn.Parameter] = [] + all_groups: List[List[nn.Parameter]] = [] + size_of_cur_group = 0 + + for param in cast(List[nn.Parameter], param_group['params']): + if size_of_cur_group + param.numel() <= max_group_size: + cur_group.append(param) + size_of_cur_group += param.numel() + else: + all_groups.append(cur_group) + cur_group = [param] + size_of_cur_group = param.numel() + + if cur_group: + all_groups.append(cur_group) + + for group in all_groups: + param_groups.append({**param_group, 'params': group}) + else: + for moe_group in group_moe.values(): + for param_group in moe_group.values(): + param_groups.append(param_group) + + return param_groups + + +def is_moe_param_group(param_group): + return param_group.get('moe', False) + + +def configure_moe_param_groups(model_parameters: List): + assert isinstance(model_parameters, list), "model_parameters must be a list" + + for p in model_parameters: + # match torch.optim.Optimizer expectations, + # see: https://github.com/pytorch/pytorch/blob/2ffab6e663b9c6951048b8c8ba82d2cc5ca5c2fc/torch/optim/optimizer.py#L270-L272 + if not isinstance(p, (torch.Tensor, dict)): + raise TypeError("param argument that would be given to the optimizer should be " + f"an iterable of Tensors or dicts, but got {type(p)}") + + # peak at the first element to determine how to proceed + first = model_parameters[0] + + # Case 1: model_parameters is a list of torch.nn.Parameter + # -> need to create moe compatible param groups + if isinstance(first, torch.nn.Parameter): + param_group = {'params': model_parameters, 'name': 'dense-params'} + return split_params_into_different_moe_groups_for_optimizer(param_group) + + # Case 2: model_parameters is a list of param groups List[dict] + # -> moe compatible param groups might already exist, if not create them + elif isinstance(first, dict): + #there are no moe groups created + if not any(['moe' in param_group for param_group in model_parameters]): + return split_params_into_different_moe_groups_for_optimizer(model_parameters) + else: + # moe groups exist, nothing to do + return model_parameters diff --git a/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/INSTALLER b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/INSTALLER new file mode 100644 index 0000000000000000000000000000000000000000..a1b589e38a32041e49332e5e81c2d363dc418d68 --- /dev/null +++ b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/INSTALLER @@ -0,0 +1 @@ +pip diff --git a/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/License.txt b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/License.txt new file mode 100644 index 0000000000000000000000000000000000000000..b491c70e0aef319022ded661e111ddbd45b8a17f --- /dev/null +++ b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/License.txt @@ -0,0 +1,1568 @@ +End User License Agreement +-------------------------- + + +Preface +------- + +The Software License Agreement in Chapter 1 and the Supplement +in Chapter 2 contain license terms and conditions that govern +the use of NVIDIA software. By accepting this agreement, you +agree to comply with all the terms and conditions applicable +to the product(s) included herein. + + +NVIDIA Driver + + +Description + +This package contains the operating system driver and +fundamental system software components for NVIDIA GPUs. + + +NVIDIA CUDA Toolkit + + +Description + +The NVIDIA CUDA Toolkit provides command-line and graphical +tools for building, debugging and optimizing the performance +of applications accelerated by NVIDIA GPUs, runtime and math +libraries, and documentation including programming guides, +user manuals, and API references. + + +Default Install Location of CUDA Toolkit + +Windows platform: + +%ProgramFiles%\NVIDIA GPU Computing Toolkit\CUDA\v#.# + +Linux platform: + +/usr/local/cuda-#.# + +Mac platform: + +/Developer/NVIDIA/CUDA-#.# + + +NVIDIA CUDA Samples + + +Description + +This package includes over 100+ CUDA examples that demonstrate +various CUDA programming principles, and efficient CUDA +implementation of algorithms in specific application domains. + + +Default Install Location of CUDA Samples + +Windows platform: + +%ProgramData%\NVIDIA Corporation\CUDA Samples\v#.# + +Linux platform: + +/usr/local/cuda-#.#/samples + +and + +$HOME/NVIDIA_CUDA-#.#_Samples + +Mac platform: + +/Developer/NVIDIA/CUDA-#.#/samples + + +NVIDIA Nsight Visual Studio Edition (Windows only) + + +Description + +NVIDIA Nsight Development Platform, Visual Studio Edition is a +development environment integrated into Microsoft Visual +Studio that provides tools for debugging, profiling, analyzing +and optimizing your GPU computing and graphics applications. + + +Default Install Location of Nsight Visual Studio Edition + +Windows platform: + +%ProgramFiles(x86)%\NVIDIA Corporation\Nsight Visual Studio Edition #.# + + +1. License Agreement for NVIDIA Software Development Kits +--------------------------------------------------------- + + +Release Date: July 26, 2018 +--------------------------- + + +Important NoticeRead before downloading, installing, +copying or using the licensed software: +------------------------------------------------------- + +This license agreement, including exhibits attached +("Agreement”) is a legal agreement between you and NVIDIA +Corporation ("NVIDIA") and governs your use of a NVIDIA +software development kit (“SDK”). + +Each SDK has its own set of software and materials, but here +is a description of the types of items that may be included in +a SDK: source code, header files, APIs, data sets and assets +(examples include images, textures, models, scenes, videos, +native API input/output files), binary software, sample code, +libraries, utility programs, programming code and +documentation. + +This Agreement can be accepted only by an adult of legal age +of majority in the country in which the SDK is used. + +If you are entering into this Agreement on behalf of a company +or other legal entity, you represent that you have the legal +authority to bind the entity to this Agreement, in which case +“you” will mean the entity you represent. + +If you don’t have the required age or authority to accept +this Agreement, or if you don’t accept all the terms and +conditions of this Agreement, do not download, install or use +the SDK. + +You agree to use the SDK only for purposes that are permitted +by (a) this Agreement, and (b) any applicable law, regulation +or generally accepted practices or guidelines in the relevant +jurisdictions. + + +1.1. License + + +1.1.1. License Grant + +Subject to the terms of this Agreement, NVIDIA hereby grants +you a non-exclusive, non-transferable license, without the +right to sublicense (except as expressly provided in this +Agreement) to: + + 1. Install and use the SDK, + + 2. Modify and create derivative works of sample source code + delivered in the SDK, and + + 3. Distribute those portions of the SDK that are identified + in this Agreement as distributable, as incorporated in + object code format into a software application that meets + the distribution requirements indicated in this Agreement. + + +1.1.2. Distribution Requirements + +These are the distribution requirements for you to exercise +the distribution grant: + + 1. Your application must have material additional + functionality, beyond the included portions of the SDK. + + 2. The distributable portions of the SDK shall only be + accessed by your application. + + 3. The following notice shall be included in modifications + and derivative works of sample source code distributed: + “This software contains source code provided by NVIDIA + Corporation.” + + 4. Unless a developer tool is identified in this Agreement + as distributable, it is delivered for your internal use + only. + + 5. The terms under which you distribute your application + must be consistent with the terms of this Agreement, + including (without limitation) terms relating to the + license grant and license restrictions and protection of + NVIDIA’s intellectual property rights. Additionally, you + agree that you will protect the privacy, security and + legal rights of your application users. + + 6. You agree to notify NVIDIA in writing of any known or + suspected distribution or use of the SDK not in compliance + with the requirements of this Agreement, and to enforce + the terms of your agreements with respect to distributed + SDK. + + +1.1.3. Authorized Users + +You may allow employees and contractors of your entity or of +your subsidiary(ies) to access and use the SDK from your +secure network to perform work on your behalf. + +If you are an academic institution you may allow users +enrolled or employed by the academic institution to access and +use the SDK from your secure network. + +You are responsible for the compliance with the terms of this +Agreement by your authorized users. If you become aware that +your authorized users didn’t follow the terms of this +Agreement, you agree to take reasonable steps to resolve the +non-compliance and prevent new occurrences. + + +1.1.4. Pre-Release SDK + +The SDK versions identified as alpha, beta, preview or +otherwise as pre-release, may not be fully functional, may +contain errors or design flaws, and may have reduced or +different security, privacy, accessibility, availability, and +reliability standards relative to commercial versions of +NVIDIA software and materials. Use of a pre-release SDK may +result in unexpected results, loss of data, project delays or +other unpredictable damage or loss. + +You may use a pre-release SDK at your own risk, understanding +that pre-release SDKs are not intended for use in production +or business-critical systems. + +NVIDIA may choose not to make available a commercial version +of any pre-release SDK. NVIDIA may also choose to abandon +development and terminate the availability of a pre-release +SDK at any time without liability. + + +1.1.5. Updates + +NVIDIA may, at its option, make available patches, workarounds +or other updates to this SDK. Unless the updates are provided +with their separate governing terms, they are deemed part of +the SDK licensed to you as provided in this Agreement. You +agree that the form and content of the SDK that NVIDIA +provides may change without prior notice to you. While NVIDIA +generally maintains compatibility between versions, NVIDIA may +in some cases make changes that introduce incompatibilities in +future versions of the SDK. + + +1.1.6. Third Party Licenses + +The SDK may come bundled with, or otherwise include or be +distributed with, third party software licensed by a NVIDIA +supplier and/or open source software provided under an open +source license. Use of third party software is subject to the +third-party license terms, or in the absence of third party +terms, the terms of this Agreement. Copyright to third party +software is held by the copyright holders indicated in the +third-party software or license. + + +1.1.7. Reservation of Rights + +NVIDIA reserves all rights, title, and interest in and to the +SDK, not expressly granted to you under this Agreement. + + +1.2. Limitations + +The following license limitations apply to your use of the +SDK: + + 1. You may not reverse engineer, decompile or disassemble, + or remove copyright or other proprietary notices from any + portion of the SDK or copies of the SDK. + + 2. Except as expressly provided in this Agreement, you may + not copy, sell, rent, sublicense, transfer, distribute, + modify, or create derivative works of any portion of the + SDK. For clarity, you may not distribute or sublicense the + SDK as a stand-alone product. + + 3. Unless you have an agreement with NVIDIA for this + purpose, you may not indicate that an application created + with the SDK is sponsored or endorsed by NVIDIA. + + 4. You may not bypass, disable, or circumvent any + encryption, security, digital rights management or + authentication mechanism in the SDK. + + 5. You may not use the SDK in any manner that would cause it + to become subject to an open source software license. As + examples, licenses that require as a condition of use, + modification, and/or distribution that the SDK be: + + a. Disclosed or distributed in source code form; + + b. Licensed for the purpose of making derivative works; + or + + c. Redistributable at no charge. + + 6. Unless you have an agreement with NVIDIA for this + purpose, you may not use the SDK with any system or + application where the use or failure of the system or + application can reasonably be expected to threaten or + result in personal injury, death, or catastrophic loss. + Examples include use in avionics, navigation, military, + medical, life support or other life critical applications. + NVIDIA does not design, test or manufacture the SDK for + these critical uses and NVIDIA shall not be liable to you + or any third party, in whole or in part, for any claims or + damages arising from such uses. + + 7. You agree to defend, indemnify and hold harmless NVIDIA + and its affiliates, and their respective employees, + contractors, agents, officers and directors, from and + against any and all claims, damages, obligations, losses, + liabilities, costs or debt, fines, restitutions and + expenses (including but not limited to attorney’s fees + and costs incident to establishing the right of + indemnification) arising out of or related to your use of + the SDK outside of the scope of this Agreement, or not in + compliance with its terms. + + +1.3. Ownership + + 1. NVIDIA or its licensors hold all rights, title and + interest in and to the SDK and its modifications and + derivative works, including their respective intellectual + property rights, subject to your rights described in this + section. This SDK may include software and materials from + NVIDIA’s licensors, and these licensors are intended + third party beneficiaries that may enforce this Agreement + with respect to their intellectual property rights. + + 2. You hold all rights, title and interest in and to your + applications and your derivative works of the sample + source code delivered in the SDK, including their + respective intellectual property rights, subject to + NVIDIA’s rights described in this section. + + 3. You may, but don’t have to, provide to NVIDIA + suggestions, feature requests or other feedback regarding + the SDK, including possible enhancements or modifications + to the SDK. For any feedback that you voluntarily provide, + you hereby grant NVIDIA and its affiliates a perpetual, + non-exclusive, worldwide, irrevocable license to use, + reproduce, modify, license, sublicense (through multiple + tiers of sublicensees), and distribute (through multiple + tiers of distributors) it without the payment of any + royalties or fees to you. NVIDIA will use feedback at its + choice. NVIDIA is constantly looking for ways to improve + its products, so you may send feedback to NVIDIA through + the developer portal at https://developer.nvidia.com. + + +1.4. No Warranties + +THE SDK IS PROVIDED BY NVIDIA “AS IS” AND “WITH ALL +FAULTS.” TO THE MAXIMUM EXTENT PERMITTED BY LAW, NVIDIA AND +ITS AFFILIATES EXPRESSLY DISCLAIM ALL WARRANTIES OF ANY KIND +OR NATURE, WHETHER EXPRESS, IMPLIED OR STATUTORY, INCLUDING, +BUT NOT LIMITED TO, ANY WARRANTIES OF MERCHANTABILITY, FITNESS +FOR A PARTICULAR PURPOSE, TITLE, NON-INFRINGEMENT, OR THE +ABSENCE OF ANY DEFECTS THEREIN, WHETHER LATENT OR PATENT. NO +WARRANTY IS MADE ON THE BASIS OF TRADE USAGE, COURSE OF +DEALING OR COURSE OF TRADE. + + +1.5. Limitation of Liability + +TO THE MAXIMUM EXTENT PERMITTED BY LAW, NVIDIA AND ITS +AFFILIATES SHALL NOT BE LIABLE FOR ANY SPECIAL, INCIDENTAL, +PUNITIVE OR CONSEQUENTIAL DAMAGES, OR ANY LOST PROFITS, LOSS +OF USE, LOSS OF DATA OR LOSS OF GOODWILL, OR THE COSTS OF +PROCURING SUBSTITUTE PRODUCTS, ARISING OUT OF OR IN CONNECTION +WITH THIS AGREEMENT OR THE USE OR PERFORMANCE OF THE SDK, +WHETHER SUCH LIABILITY ARISES FROM ANY CLAIM BASED UPON BREACH +OF CONTRACT, BREACH OF WARRANTY, TORT (INCLUDING NEGLIGENCE), +PRODUCT LIABILITY OR ANY OTHER CAUSE OF ACTION OR THEORY OF +LIABILITY. IN NO EVENT WILL NVIDIA’S AND ITS AFFILIATES +TOTAL CUMULATIVE LIABILITY UNDER OR ARISING OUT OF THIS +AGREEMENT EXCEED US$10.00. THE NATURE OF THE LIABILITY OR THE +NUMBER OF CLAIMS OR SUITS SHALL NOT ENLARGE OR EXTEND THIS +LIMIT. + +These exclusions and limitations of liability shall apply +regardless if NVIDIA or its affiliates have been advised of +the possibility of such damages, and regardless of whether a +remedy fails its essential purpose. These exclusions and +limitations of liability form an essential basis of the +bargain between the parties, and, absent any of these +exclusions or limitations of liability, the provisions of this +Agreement, including, without limitation, the economic terms, +would be substantially different. + + +1.6. Termination + + 1. This Agreement will continue to apply until terminated by + either you or NVIDIA as described below. + + 2. If you want to terminate this Agreement, you may do so by + stopping to use the SDK. + + 3. NVIDIA may, at any time, terminate this Agreement if: + + a. (i) you fail to comply with any term of this + Agreement and the non-compliance is not fixed within + thirty (30) days following notice from NVIDIA (or + immediately if you violate NVIDIA’s intellectual + property rights); + + b. (ii) you commence or participate in any legal + proceeding against NVIDIA with respect to the SDK; or + + c. (iii) NVIDIA decides to no longer provide the SDK in + a country or, in NVIDIA’s sole discretion, the + continued use of it is no longer commercially viable. + + 4. Upon any termination of this Agreement, you agree to + promptly discontinue use of the SDK and destroy all copies + in your possession or control. Your prior distributions in + accordance with this Agreement are not affected by the + termination of this Agreement. Upon written request, you + will certify in writing that you have complied with your + commitments under this section. Upon any termination of + this Agreement all provisions survive except for the + license grant provisions. + + +1.7. General + +If you wish to assign this Agreement or your rights and +obligations, including by merger, consolidation, dissolution +or operation of law, contact NVIDIA to ask for permission. Any +attempted assignment not approved by NVIDIA in writing shall +be void and of no effect. NVIDIA may assign, delegate or +transfer this Agreement and its rights and obligations, and if +to a non-affiliate you will be notified. + +You agree to cooperate with NVIDIA and provide reasonably +requested information to verify your compliance with this +Agreement. + +This Agreement will be governed in all respects by the laws of +the United States and of the State of Delaware as those laws +are applied to contracts entered into and performed entirely +within Delaware by Delaware residents, without regard to the +conflicts of laws principles. The United Nations Convention on +Contracts for the International Sale of Goods is specifically +disclaimed. You agree to all terms of this Agreement in the +English language. + +The state or federal courts residing in Santa Clara County, +California shall have exclusive jurisdiction over any dispute +or claim arising out of this Agreement. Notwithstanding this, +you agree that NVIDIA shall still be allowed to apply for +injunctive remedies or an equivalent type of urgent legal +relief in any jurisdiction. + +If any court of competent jurisdiction determines that any +provision of this Agreement is illegal, invalid or +unenforceable, such provision will be construed as limited to +the extent necessary to be consistent with and fully +enforceable under the law and the remaining provisions will +remain in full force and effect. Unless otherwise specified, +remedies are cumulative. + +Each party acknowledges and agrees that the other is an +independent contractor in the performance of this Agreement. + +The SDK has been developed entirely at private expense and is +“commercial items” consisting of “commercial computer +software” and “commercial computer software +documentation” provided with RESTRICTED RIGHTS. Use, +duplication or disclosure by the U.S. Government or a U.S. +Government subcontractor is subject to the restrictions in +this Agreement pursuant to DFARS 227.7202-3(a) or as set forth +in subparagraphs (c)(1) and (2) of the Commercial Computer +Software - Restricted Rights clause at FAR 52.227-19, as +applicable. Contractor/manufacturer is NVIDIA, 2788 San Tomas +Expressway, Santa Clara, CA 95051. + +The SDK is subject to United States export laws and +regulations. You agree that you will not ship, transfer or +export the SDK into any country, or use the SDK in any manner, +prohibited by the United States Bureau of Industry and +Security or economic sanctions regulations administered by the +U.S. Department of Treasury’s Office of Foreign Assets +Control (OFAC), or any applicable export laws, restrictions or +regulations. These laws include restrictions on destinations, +end users and end use. By accepting this Agreement, you +confirm that you are not a resident or citizen of any country +currently embargoed by the U.S. and that you are not otherwise +prohibited from receiving the SDK. + +Any notice delivered by NVIDIA to you under this Agreement +will be delivered via mail, email or fax. You agree that any +notices that NVIDIA sends you electronically will satisfy any +legal communication requirements. Please direct your legal +notices or other correspondence to NVIDIA Corporation, 2788 +San Tomas Expressway, Santa Clara, California 95051, United +States of America, Attention: Legal Department. + +This Agreement and any exhibits incorporated into this +Agreement constitute the entire agreement of the parties with +respect to the subject matter of this Agreement and supersede +all prior negotiations or documentation exchanged between the +parties relating to this SDK license. Any additional and/or +conflicting terms on documents issued by you are null, void, +and invalid. Any amendment or waiver under this Agreement +shall be in writing and signed by representatives of both +parties. + + +2. CUDA Toolkit Supplement to Software License Agreement for +NVIDIA Software Development Kits +------------------------------------------------------------ + + +Release date: August 16, 2018 +----------------------------- + +The terms in this supplement govern your use of the NVIDIA +CUDA Toolkit SDK under the terms of your license agreement +(“Agreement”) as modified by this supplement. Capitalized +terms used but not defined below have the meaning assigned to +them in the Agreement. + +This supplement is an exhibit to the Agreement and is +incorporated as an integral part of the Agreement. In the +event of conflict between the terms in this supplement and the +terms in the Agreement, the terms in this supplement govern. + + +2.1. License Scope + +The SDK is licensed for you to develop applications only for +use in systems with NVIDIA GPUs. + + +2.2. Distribution + +The portions of the SDK that are distributable under the +Agreement are listed in Attachment A. + + +2.3. Operating Systems + +Those portions of the SDK designed exclusively for use on the +Linux or FreeBSD operating systems, or other operating systems +derived from the source code to these operating systems, may +be copied and redistributed for use in accordance with this +Agreement, provided that the object code files are not +modified in any way (except for unzipping of compressed +files). + + +2.4. Audio and Video Encoders and Decoders + +You acknowledge and agree that it is your sole responsibility +to obtain any additional third-party licenses required to +make, have made, use, have used, sell, import, and offer for +sale your products or services that include or incorporate any +third-party software and content relating to audio and/or +video encoders and decoders from, including but not limited +to, Microsoft, Thomson, Fraunhofer IIS, Sisvel S.p.A., +MPEG-LA, and Coding Technologies. NVIDIA does not grant to you +under this Agreement any necessary patent or other rights with +respect to any audio and/or video encoders and decoders. + + +2.5. Licensing + +If the distribution terms in this Agreement are not suitable +for your organization, or for any questions regarding this +Agreement, please contact NVIDIA at +nvidia-compute-license-questions@nvidia.com. + + +2.6. Attachment A + +The following portions of the SDK are distributable under the +Agreement: + +Component + +CUDA Runtime + +Windows + +cudart.dll, cudart_static.lib, cudadevrt.lib + +Mac OSX + +libcudart.dylib, libcudart_static.a, libcudadevrt.a + +Linux + +libcudart.so, libcudart_static.a, libcudadevrt.a + +Android + +libcudart.so, libcudart_static.a, libcudadevrt.a + +Component + +CUDA FFT Library + +Windows + +cufft.dll, cufftw.dll, cufft.lib, cufftw.lib + +Mac OSX + +libcufft.dylib, libcufft_static.a, libcufftw.dylib, +libcufftw_static.a + +Linux + +libcufft.so, libcufft_static.a, libcufftw.so, +libcufftw_static.a + +Android + +libcufft.so, libcufft_static.a, libcufftw.so, +libcufftw_static.a + +Component + +CUDA BLAS Library + +Windows + +cublas.dll, cublasLt.dll + +Mac OSX + +libcublas.dylib, libcublasLt.dylib, libcublas_static.a, +libcublasLt_static.a + +Linux + +libcublas.so, libcublasLt.so, libcublas_static.a, +libcublasLt_static.a + +Android + +libcublas.so, libcublasLt.so, libcublas_static.a, +libcublasLt_static.a + +Component + +NVIDIA "Drop-in" BLAS Library + +Windows + +nvblas.dll + +Mac OSX + +libnvblas.dylib + +Linux + +libnvblas.so + +Component + +CUDA Sparse Matrix Library + +Windows + +cusparse.dll, cusparse.lib + +Mac OSX + +libcusparse.dylib, libcusparse_static.a + +Linux + +libcusparse.so, libcusparse_static.a + +Android + +libcusparse.so, libcusparse_static.a + +Component + +CUDA Linear Solver Library + +Windows + +cusolver.dll, cusolver.lib + +Mac OSX + +libcusolver.dylib, libcusolver_static.a + +Linux + +libcusolver.so, libcusolver_static.a + +Android + +libcusolver.so, libcusolver_static.a + +Component + +CUDA Random Number Generation Library + +Windows + +curand.dll, curand.lib + +Mac OSX + +libcurand.dylib, libcurand_static.a + +Linux + +libcurand.so, libcurand_static.a + +Android + +libcurand.so, libcurand_static.a + +Component + +CUDA Accelerated Graph Library + +Component + +NVIDIA Performance Primitives Library + +Windows + +nppc.dll, nppc.lib, nppial.dll, nppial.lib, nppicc.dll, +nppicc.lib, nppicom.dll, nppicom.lib, nppidei.dll, +nppidei.lib, nppif.dll, nppif.lib, nppig.dll, nppig.lib, +nppim.dll, nppim.lib, nppist.dll, nppist.lib, nppisu.dll, +nppisu.lib, nppitc.dll, nppitc.lib, npps.dll, npps.lib + +Mac OSX + +libnppc.dylib, libnppc_static.a, libnppial.dylib, +libnppial_static.a, libnppicc.dylib, libnppicc_static.a, +libnppicom.dylib, libnppicom_static.a, libnppidei.dylib, +libnppidei_static.a, libnppif.dylib, libnppif_static.a, +libnppig.dylib, libnppig_static.a, libnppim.dylib, +libnppisu_static.a, libnppitc.dylib, libnppitc_static.a, +libnpps.dylib, libnpps_static.a + +Linux + +libnppc.so, libnppc_static.a, libnppial.so, +libnppial_static.a, libnppicc.so, libnppicc_static.a, +libnppicom.so, libnppicom_static.a, libnppidei.so, +libnppidei_static.a, libnppif.so, libnppif_static.a +libnppig.so, libnppig_static.a, libnppim.so, +libnppim_static.a, libnppist.so, libnppist_static.a, +libnppisu.so, libnppisu_static.a, libnppitc.so +libnppitc_static.a, libnpps.so, libnpps_static.a + +Android + +libnppc.so, libnppc_static.a, libnppial.so, +libnppial_static.a, libnppicc.so, libnppicc_static.a, +libnppicom.so, libnppicom_static.a, libnppidei.so, +libnppidei_static.a, libnppif.so, libnppif_static.a +libnppig.so, libnppig_static.a, libnppim.so, +libnppim_static.a, libnppist.so, libnppist_static.a, +libnppisu.so, libnppisu_static.a, libnppitc.so +libnppitc_static.a, libnpps.so, libnpps_static.a + +Component + +NVIDIA JPEG Library + +Linux + +libnvjpeg.so, libnvjpeg_static.a + +Component + +Internal common library required for statically linking to +cuBLAS, cuSPARSE, cuFFT, cuRAND, nvJPEG and NPP + +Mac OSX + +libculibos.a + +Linux + +libculibos.a + +Component + +NVIDIA Runtime Compilation Library and Header + +All + +nvrtc.h + +Windows + +nvrtc.dll, nvrtc-builtins.dll + +Mac OSX + +libnvrtc.dylib, libnvrtc-builtins.dylib + +Linux + +libnvrtc.so, libnvrtc-builtins.so + +Component + +NVIDIA Optimizing Compiler Library + +Windows + +nvvm.dll + +Mac OSX + +libnvvm.dylib + +Linux + +libnvvm.so + +Component + +NVIDIA Common Device Math Functions Library + +Windows + +libdevice.10.bc + +Mac OSX + +libdevice.10.bc + +Linux + +libdevice.10.bc + +Component + +CUDA Occupancy Calculation Header Library + +All + +cuda_occupancy.h + +Component + +CUDA Half Precision Headers + +All + +cuda_fp16.h, cuda_fp16.hpp + +Component + +CUDA Profiling Tools Interface (CUPTI) Library + +Windows + +cupti.dll + +Mac OSX + +libcupti.dylib + +Linux + +libcupti.so + +Component + +NVIDIA Tools Extension Library + +Windows + +nvToolsExt.dll, nvToolsExt.lib + +Mac OSX + +libnvToolsExt.dylib + +Linux + +libnvToolsExt.so + +Component + +NVIDIA CUDA Driver Libraries + +Linux + +libcuda.so, libnvidia-fatbinaryloader.so, +libnvidia-ptxjitcompiler.so + +The NVIDIA CUDA Driver Libraries are only distributable in +applications that meet this criteria: + + 1. The application was developed starting from a NVIDIA CUDA + container obtained from Docker Hub or the NVIDIA GPU + Cloud, and + + 2. The resulting application is packaged as a Docker + container and distributed to users on Docker Hub or the + NVIDIA GPU Cloud only. + + +2.7. Attachment B + + +Additional Licensing Obligations + +The following third party components included in the SOFTWARE +are licensed to Licensee pursuant to the following terms and +conditions: + + 1. Licensee's use of the GDB third party component is + subject to the terms and conditions of GNU GPL v3: + + This product includes copyrighted third-party software licensed + under the terms of the GNU General Public License v3 ("GPL v3"). + All third-party software packages are copyright by their respective + authors. GPL v3 terms and conditions are hereby incorporated into + the Agreement by this reference: http://www.gnu.org/licenses/gpl.txt + + Consistent with these licensing requirements, the software + listed below is provided under the terms of the specified + open source software licenses. To obtain source code for + software provided under licenses that require + redistribution of source code, including the GNU General + Public License (GPL) and GNU Lesser General Public License + (LGPL), contact oss-requests@nvidia.com. This offer is + valid for a period of three (3) years from the date of the + distribution of this product by NVIDIA CORPORATION. + + Component License + CUDA-GDB GPL v3 + + 2. Licensee represents and warrants that any and all third + party licensing and/or royalty payment obligations in + connection with Licensee's use of the H.264 video codecs + are solely the responsibility of Licensee. + + 3. Licensee's use of the Thrust library is subject to the + terms and conditions of the Apache License Version 2.0. + All third-party software packages are copyright by their + respective authors. Apache License Version 2.0 terms and + conditions are hereby incorporated into the Agreement by + this reference. + http://www.apache.org/licenses/LICENSE-2.0.html + + In addition, Licensee acknowledges the following notice: + Thrust includes source code from the Boost Iterator, + Tuple, System, and Random Number libraries. + + Boost Software License - Version 1.0 - August 17th, 2003 + . . . . + + Permission is hereby granted, free of charge, to any person or + organization obtaining a copy of the software and accompanying + documentation covered by this license (the "Software") to use, + reproduce, display, distribute, execute, and transmit the Software, + and to prepare derivative works of the Software, and to permit + third-parties to whom the Software is furnished to do so, all + subject to the following: + + The copyright notices in the Software and this entire statement, + including the above license grant, this restriction and the following + disclaimer, must be included in all copies of the Software, in whole + or in part, and all derivative works of the Software, unless such + copies or derivative works are solely in the form of machine-executable + object code generated by a source language processor. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF + MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, TITLE AND + NON-INFRINGEMENT. IN NO EVENT SHALL THE COPYRIGHT HOLDERS OR + ANYONE DISTRIBUTING THE SOFTWARE BE LIABLE FOR ANY DAMAGES OR + OTHER LIABILITY, WHETHER IN CONTRACT, TORT OR OTHERWISE, ARISING + FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR + OTHER DEALINGS IN THE SOFTWARE. + + 4. Licensee's use of the LLVM third party component is + subject to the following terms and conditions: + + ====================================================== + LLVM Release License + ====================================================== + University of Illinois/NCSA + Open Source License + + Copyright (c) 2003-2010 University of Illinois at Urbana-Champaign. + All rights reserved. + + Developed by: + + LLVM Team + + University of Illinois at Urbana-Champaign + + http://llvm.org + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to + deal with the Software without restriction, including without limitation the + rights to use, copy, modify, merge, publish, distribute, sublicense, and/or + sell copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + * Redistributions of source code must retain the above copyright notice, + this list of conditions and the following disclaimers. + + * Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimers in the + documentation and/or other materials provided with the distribution. + + * Neither the names of the LLVM Team, University of Illinois at Urbana- + Champaign, nor the names of its contributors may be used to endorse or + promote products derived from this Software without specific prior + written permission. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL + THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR + OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, + ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER + DEALINGS WITH THE SOFTWARE. + + 5. Licensee's use (e.g. nvprof) of the PCRE third party + component is subject to the following terms and + conditions: + + ------------ + PCRE LICENCE + ------------ + PCRE is a library of functions to support regular expressions whose syntax + and semantics are as close as possible to those of the Perl 5 language. + Release 8 of PCRE is distributed under the terms of the "BSD" licence, as + specified below. The documentation for PCRE, supplied in the "doc" + directory, is distributed under the same terms as the software itself. The + basic library functions are written in C and are freestanding. Also + included in the distribution is a set of C++ wrapper functions, and a just- + in-time compiler that can be used to optimize pattern matching. These are + both optional features that can be omitted when the library is built. + + THE BASIC LIBRARY FUNCTIONS + --------------------------- + Written by: Philip Hazel + Email local part: ph10 + Email domain: cam.ac.uk + University of Cambridge Computing Service, + Cambridge, England. + Copyright (c) 1997-2012 University of Cambridge + All rights reserved. + + PCRE JUST-IN-TIME COMPILATION SUPPORT + ------------------------------------- + Written by: Zoltan Herczeg + Email local part: hzmester + Emain domain: freemail.hu + Copyright(c) 2010-2012 Zoltan Herczeg + All rights reserved. + + STACK-LESS JUST-IN-TIME COMPILER + -------------------------------- + Written by: Zoltan Herczeg + Email local part: hzmester + Emain domain: freemail.hu + Copyright(c) 2009-2012 Zoltan Herczeg + All rights reserved. + + THE C++ WRAPPER FUNCTIONS + ------------------------- + Contributed by: Google Inc. + Copyright (c) 2007-2012, Google Inc. + All rights reserved. + + THE "BSD" LICENCE + ----------------- + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are met: + + * Redistributions of source code must retain the above copyright notice, + this list of conditions and the following disclaimer. + + * Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + * Neither the name of the University of Cambridge nor the name of Google + Inc. nor the names of their contributors may be used to endorse or + promote products derived from this software without specific prior + written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" + AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE + IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE + ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE + LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR + CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF + SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS + INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN + CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) + ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE + POSSIBILITY OF SUCH DAMAGE. + + 6. Some of the cuBLAS library routines were written by or + derived from code written by Vasily Volkov and are subject + to the Modified Berkeley Software Distribution License as + follows: + + Copyright (c) 2007-2009, Regents of the University of California + + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are + met: + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + * Redistributions in binary form must reproduce the above + copyright notice, this list of conditions and the following + disclaimer in the documentation and/or other materials provided + with the distribution. + * Neither the name of the University of California, Berkeley nor + the names of its contributors may be used to endorse or promote + products derived from this software without specific prior + written permission. + + THIS SOFTWARE IS PROVIDED BY THE AUTHOR "AS IS" AND ANY EXPRESS OR + IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + DISCLAIMED. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY DIRECT, + INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR + SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) + HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, + STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING + IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE + POSSIBILITY OF SUCH DAMAGE. + + 7. Some of the cuBLAS library routines were written by or + derived from code written by Davide Barbieri and are + subject to the Modified Berkeley Software Distribution + License as follows: + + Copyright (c) 2008-2009 Davide Barbieri @ University of Rome Tor Vergata. + + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are + met: + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + * Redistributions in binary form must reproduce the above + copyright notice, this list of conditions and the following + disclaimer in the documentation and/or other materials provided + with the distribution. + * The name of the author may not be used to endorse or promote + products derived from this software without specific prior + written permission. + + THIS SOFTWARE IS PROVIDED BY THE AUTHOR "AS IS" AND ANY EXPRESS OR + IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED + WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE + DISCLAIMED. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY DIRECT, + INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES + (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR + SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) + HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, + STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING + IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE + POSSIBILITY OF SUCH DAMAGE. + + 8. Some of the cuBLAS library routines were derived from + code developed by the University of Tennessee and are + subject to the Modified Berkeley Software Distribution + License as follows: + + Copyright (c) 2010 The University of Tennessee. + + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are + met: + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + * Redistributions in binary form must reproduce the above + copyright notice, this list of conditions and the following + disclaimer listed in this license in the documentation and/or + other materials provided with the distribution. + * Neither the name of the copyright holders nor the names of its + contributors may be used to endorse or promote products derived + from this software without specific prior written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT + OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT + LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, + DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY + THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + 9. Some of the cuBLAS library routines were written by or + derived from code written by Jonathan Hogg and are subject + to the Modified Berkeley Software Distribution License as + follows: + + Copyright (c) 2012, The Science and Technology Facilities Council (STFC). + + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are + met: + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + * Redistributions in binary form must reproduce the above + copyright notice, this list of conditions and the following + disclaimer in the documentation and/or other materials provided + with the distribution. + * Neither the name of the STFC nor the names of its contributors + may be used to endorse or promote products derived from this + software without specific prior written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE STFC BE + LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR + CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF + SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR + BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, + WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE + OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN + IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + 10. Some of the cuBLAS library routines were written by or + derived from code written by Ahmad M. Abdelfattah, David + Keyes, and Hatem Ltaief, and are subject to the Apache + License, Version 2.0, as follows: + + -- (C) Copyright 2013 King Abdullah University of Science and Technology + Authors: + Ahmad Abdelfattah (ahmad.ahmad@kaust.edu.sa) + David Keyes (david.keyes@kaust.edu.sa) + Hatem Ltaief (hatem.ltaief@kaust.edu.sa) + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + * Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + * Neither the name of the King Abdullah University of Science and + Technology nor the names of its contributors may be used to endorse + or promote products derived from this software without specific prior + written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT + HOLDERS OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT + LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, + DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY + THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE + + 11. Some of the cuSPARSE library routines were written by or + derived from code written by Li-Wen Chang and are subject + to the NCSA Open Source License as follows: + + Copyright (c) 2012, University of Illinois. + + All rights reserved. + + Developed by: IMPACT Group, University of Illinois, http://impact.crhc.illinois.edu + + Permission is hereby granted, free of charge, to any person obtaining + a copy of this software and associated documentation files (the + "Software"), to deal with the Software without restriction, including + without limitation the rights to use, copy, modify, merge, publish, + distribute, sublicense, and/or sell copies of the Software, and to + permit persons to whom the Software is furnished to do so, subject to + the following conditions: + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + * Redistributions in binary form must reproduce the above + copyright notice, this list of conditions and the following + disclaimers in the documentation and/or other materials provided + with the distribution. + * Neither the names of IMPACT Group, University of Illinois, nor + the names of its contributors may be used to endorse or promote + products derived from this Software without specific prior + written permission. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF + MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND + NONINFRINGEMENT. IN NO EVENT SHALL THE CONTRIBUTORS OR COPYRIGHT + HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER + IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR + IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS WITH THE + SOFTWARE. + + 12. Some of the cuRAND library routines were written by or + derived from code written by Mutsuo Saito and Makoto + Matsumoto and are subject to the following license: + + Copyright (c) 2009, 2010 Mutsuo Saito, Makoto Matsumoto and Hiroshima + University. All rights reserved. + + Copyright (c) 2011 Mutsuo Saito, Makoto Matsumoto, Hiroshima + University and University of Tokyo. All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are + met: + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + * Redistributions in binary form must reproduce the above + copyright notice, this list of conditions and the following + disclaimer in the documentation and/or other materials provided + with the distribution. + * Neither the name of the Hiroshima University nor the names of + its contributors may be used to endorse or promote products + derived from this software without specific prior written + permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT + OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT + LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, + DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY + THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + 13. Some of the cuRAND library routines were derived from + code developed by D. E. Shaw Research and are subject to + the following license: + + Copyright 2010-2011, D. E. Shaw Research. + + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are + met: + * Redistributions of source code must retain the above copyright + notice, this list of conditions, and the following disclaimer. + * Redistributions in binary form must reproduce the above + copyright notice, this list of conditions, and the following + disclaimer in the documentation and/or other materials provided + with the distribution. + * Neither the name of D. E. Shaw Research nor the names of its + contributors may be used to endorse or promote products derived + from this software without specific prior written permission. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT + OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT + LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, + DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY + THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + 14. Some of the Math library routines were written by or + derived from code developed by Norbert Juffa and are + subject to the following license: + + Copyright (c) 2015-2017, Norbert Juffa + All rights reserved. + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + 1. Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + 2. Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT + HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT + LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, + DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY + THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + 15. Licensee's use of the lz4 third party component is + subject to the following terms and conditions: + + Copyright (C) 2011-2013, Yann Collet. + BSD 2-Clause License (http://www.opensource.org/licenses/bsd-license.php) + + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions are + met: + + * Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + * Redistributions in binary form must reproduce the above + copyright notice, this list of conditions and the following disclaimer + in the documentation and/or other materials provided with the + distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT + OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, + SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT + LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, + DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY + THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT + (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE + OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. + + 16. The NPP library uses code from the Boost Math Toolkit, + and is subject to the following license: + + Boost Software License - Version 1.0 - August 17th, 2003 + . . . . + + Permission is hereby granted, free of charge, to any person or + organization obtaining a copy of the software and accompanying + documentation covered by this license (the "Software") to use, + reproduce, display, distribute, execute, and transmit the Software, + and to prepare derivative works of the Software, and to permit + third-parties to whom the Software is furnished to do so, all + subject to the following: + + The copyright notices in the Software and this entire statement, + including the above license grant, this restriction and the following + disclaimer, must be included in all copies of the Software, in whole + or in part, and all derivative works of the Software, unless such + copies or derivative works are solely in the form of machine-executable + object code generated by a source language processor. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF + MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, TITLE AND + NON-INFRINGEMENT. IN NO EVENT SHALL THE COPYRIGHT HOLDERS OR + ANYONE DISTRIBUTING THE SOFTWARE BE LIABLE FOR ANY DAMAGES OR + OTHER LIABILITY, WHETHER IN CONTRACT, TORT OR OTHERWISE, ARISING + FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR + OTHER DEALINGS IN THE SOFTWARE. + + 17. Portions of the Nsight Eclipse Edition is subject to the + following license: + + The Eclipse Foundation makes available all content in this plug-in + ("Content"). Unless otherwise indicated below, the Content is provided + to you under the terms and conditions of the Eclipse Public License + Version 1.0 ("EPL"). A copy of the EPL is available at http:// + www.eclipse.org/legal/epl-v10.html. For purposes of the EPL, "Program" + will mean the Content. + + If you did not receive this Content directly from the Eclipse + Foundation, the Content is being redistributed by another party + ("Redistributor") and different terms and conditions may apply to your + use of any object code in the Content. Check the Redistributor's + license that was provided with the Content. If no such license exists, + contact the Redistributor. Unless otherwise indicated below, the terms + and conditions of the EPL still apply to any source code in the + Content and such source code may be obtained at http://www.eclipse.org. + + 18. Some of the cuBLAS library routines uses code from + OpenAI, which is subject to the following license: + + License URL + https://github.com/openai/openai-gemm/blob/master/LICENSE + + License Text + The MIT License + + Copyright (c) 2016 OpenAI (http://openai.com), 2016 Google Inc. + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in + all copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN + THE SOFTWARE. + + 19. Licensee's use of the Visual Studio Setup Configuration + Samples is subject to the following license: + + The MIT License (MIT) + Copyright (C) Microsoft Corporation. All rights reserved. + + Permission is hereby granted, free of charge, to any person + obtaining a copy of this software and associated documentation + files (the "Software"), to deal in the Software without restriction, + including without limitation the rights to use, copy, modify, merge, + publish, distribute, sublicense, and/or sell copies of the Software, + and to permit persons to whom the Software is furnished to do so, + subject to the following conditions: + + The above copyright notice and this permission notice shall be included + in all copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS + OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + + 20. Licensee's use of linmath.h header for CPU functions for + GL vector/matrix operations from lunarG is subject to the + Apache License Version 2.0. + + 21. The DX12-CUDA sample uses the d3dx12.h header, which is + subject to the MIT license . + +----------------- diff --git a/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/METADATA b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/METADATA new file mode 100644 index 0000000000000000000000000000000000000000..7f3d3db95ffbbe222c91c7284524edc471dce14f --- /dev/null +++ b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/METADATA @@ -0,0 +1,35 @@ +Metadata-Version: 2.1 +Name: nvidia-cufft-cu12 +Version: 11.2.1.3 +Summary: CUFFT native runtime libraries +Home-page: https://developer.nvidia.com/cuda-zone +Author: Nvidia CUDA Installer Team +Author-email: cuda_installer@nvidia.com +License: NVIDIA Proprietary Software +Keywords: cuda,nvidia,runtime,machine learning,deep learning +Classifier: Development Status :: 4 - Beta +Classifier: Intended Audience :: Developers +Classifier: Intended Audience :: Education +Classifier: Intended Audience :: Science/Research +Classifier: License :: Other/Proprietary License +Classifier: Natural Language :: English +Classifier: Programming Language :: Python :: 3 +Classifier: Programming Language :: Python :: 3.5 +Classifier: Programming Language :: Python :: 3.6 +Classifier: Programming Language :: Python :: 3.7 +Classifier: Programming Language :: Python :: 3.8 +Classifier: Programming Language :: Python :: 3.9 +Classifier: Programming Language :: Python :: 3.10 +Classifier: Programming Language :: Python :: 3.11 +Classifier: Programming Language :: Python :: 3 :: Only +Classifier: Topic :: Scientific/Engineering +Classifier: Topic :: Scientific/Engineering :: Mathematics +Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence +Classifier: Topic :: Software Development +Classifier: Topic :: Software Development :: Libraries +Classifier: Operating System :: Microsoft :: Windows +Classifier: Operating System :: POSIX :: Linux +Requires-Python: >=3 +License-File: License.txt + +CUFFT native runtime libraries diff --git a/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/RECORD b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/RECORD new file mode 100644 index 0000000000000000000000000000000000000000..b18ccb7914cb4adc60564cdaaa36cdb566059576 --- /dev/null +++ b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/RECORD @@ -0,0 +1,21 @@ +nvidia/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0 +nvidia/__pycache__/__init__.cpython-312.pyc,, +nvidia/cufft/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0 +nvidia/cufft/__pycache__/__init__.cpython-312.pyc,, +nvidia/cufft/include/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0 +nvidia/cufft/include/__pycache__/__init__.cpython-312.pyc,, +nvidia/cufft/include/cudalibxt.h,sha256=9GDuRiOzJuO61zRDhIpWpF7XHp8FXSOIlHJNoIMwOZQ,4105 +nvidia/cufft/include/cufft.h,sha256=OPTrbN3YvHR2HZTy4Kr_azbFUz8ZGXAkmT_1ero1y3I,13109 +nvidia/cufft/include/cufftXt.h,sha256=bTMo9ixYPn-FnrCw2VYZ2XVwDYT7N8WrRdXp4CmBilY,11148 +nvidia/cufft/include/cufftw.h,sha256=Uzfj1IVMlLQU_G50u84hXYX1K95HLXIwOcjQoAg5pGE,20051 +nvidia/cufft/lib/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0 +nvidia/cufft/lib/__pycache__/__init__.cpython-312.pyc,, +nvidia/cufft/lib/libcufft.so.11,sha256=85IcQTOSUkJFnr_b95AdtOv65rvP_53FzlOx_xP7Qv8,292889192 +nvidia/cufft/lib/libcufftw.so.11,sha256=IwelrPzMm0D5iThAOCGM_q1WTNQ2M3AdMMiTBH50T0Q,974888 +nvidia_cufft_cu12-11.2.1.3.dist-info/INSTALLER,sha256=zuuue4knoyJ-UwPPXg8fezS7VCrXJQrAP7zeNuwvFQg,4 +nvidia_cufft_cu12-11.2.1.3.dist-info/License.txt,sha256=rW9YU_ugyg0VnQ9Y1JrkmDDC-Mk_epJki5zpCttMbM0,59262 +nvidia_cufft_cu12-11.2.1.3.dist-info/METADATA,sha256=e3c8JR1hTBAIlY96lfSibInmFGkkwNcYO6CExGuXQ6w,1502 +nvidia_cufft_cu12-11.2.1.3.dist-info/RECORD,, +nvidia_cufft_cu12-11.2.1.3.dist-info/REQUESTED,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0 +nvidia_cufft_cu12-11.2.1.3.dist-info/WHEEL,sha256=XDTs3wIbcE-BcRO08VJlZpA6z9OaC1mOKPCGGGwuM2g,109 +nvidia_cufft_cu12-11.2.1.3.dist-info/top_level.txt,sha256=fTkAtiFuL16nUrB9ytDDtpytz2t0B4NvYTnRzwAhO14,7 diff --git a/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/REQUESTED b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/REQUESTED new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/WHEEL b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/WHEEL new file mode 100644 index 0000000000000000000000000000000000000000..e6c30e957cfb045017a9fef3430bb8ee87c4a074 --- /dev/null +++ b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/WHEEL @@ -0,0 +1,5 @@ +Wheel-Version: 1.0 +Generator: bdist_wheel (0.42.0) +Root-Is-Purelib: true +Tag: py3-none-manylinux2014_x86_64 + diff --git a/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/top_level.txt b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/top_level.txt new file mode 100644 index 0000000000000000000000000000000000000000..862f7abf232cdfbb928609856247292e81c9decb --- /dev/null +++ b/lib/python3.12/site-packages/nvidia_cufft_cu12-11.2.1.3.dist-info/top_level.txt @@ -0,0 +1 @@ +nvidia diff --git a/lib/python3.12/site-packages/pygments-2.19.2.dist-info/INSTALLER b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/INSTALLER new file mode 100644 index 0000000000000000000000000000000000000000..a1b589e38a32041e49332e5e81c2d363dc418d68 --- /dev/null +++ b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/INSTALLER @@ -0,0 +1 @@ +pip diff --git a/lib/python3.12/site-packages/pygments-2.19.2.dist-info/METADATA b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/METADATA new file mode 100644 index 0000000000000000000000000000000000000000..2eff6a0c3865f050611859a41df6bf81aaf4f88e --- /dev/null +++ b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/METADATA @@ -0,0 +1,58 @@ +Metadata-Version: 2.4 +Name: Pygments +Version: 2.19.2 +Summary: Pygments is a syntax highlighting package written in Python. +Project-URL: Homepage, https://pygments.org +Project-URL: Documentation, https://pygments.org/docs +Project-URL: Source, https://github.com/pygments/pygments +Project-URL: Bug Tracker, https://github.com/pygments/pygments/issues +Project-URL: Changelog, https://github.com/pygments/pygments/blob/master/CHANGES +Author-email: Georg Brandl +Maintainer: Matthäus G. Chajdas +Maintainer-email: Georg Brandl , Jean Abou Samra +License: BSD-2-Clause +License-File: AUTHORS +License-File: LICENSE +Keywords: syntax highlighting +Classifier: Development Status :: 6 - Mature +Classifier: Intended Audience :: Developers +Classifier: Intended Audience :: End Users/Desktop +Classifier: Intended Audience :: System Administrators +Classifier: License :: OSI Approved :: BSD License +Classifier: Operating System :: OS Independent +Classifier: Programming Language :: Python +Classifier: Programming Language :: Python :: 3 +Classifier: Programming Language :: Python :: 3.8 +Classifier: Programming Language :: Python :: 3.9 +Classifier: Programming Language :: Python :: 3.10 +Classifier: Programming Language :: Python :: 3.11 +Classifier: Programming Language :: Python :: 3.12 +Classifier: Programming Language :: Python :: 3.13 +Classifier: Programming Language :: Python :: Implementation :: CPython +Classifier: Programming Language :: Python :: Implementation :: PyPy +Classifier: Topic :: Text Processing :: Filters +Classifier: Topic :: Utilities +Requires-Python: >=3.8 +Provides-Extra: plugins +Provides-Extra: windows-terminal +Requires-Dist: colorama>=0.4.6; extra == 'windows-terminal' +Description-Content-Type: text/x-rst + +Pygments +~~~~~~~~ + +Pygments is a syntax highlighting package written in Python. + +It is a generic syntax highlighter suitable for use in code hosting, forums, +wikis or other applications that need to prettify source code. Highlights +are: + +* a wide range of over 500 languages and other text formats is supported +* special attention is paid to details, increasing quality by a fair amount +* support for new languages and formats are added easily +* a number of output formats, presently HTML, LaTeX, RTF, SVG, all image + formats that PIL supports and ANSI sequences +* it is usable as a command-line tool and as a library + +Copyright 2006-2025 by the Pygments team, see ``AUTHORS``. +Licensed under the BSD, see ``LICENSE`` for details. diff --git a/lib/python3.12/site-packages/pygments-2.19.2.dist-info/RECORD b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/RECORD new file mode 100644 index 0000000000000000000000000000000000000000..40a502bca3f0bb57a455b5fac56008a30d63ba23 --- /dev/null +++ b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/RECORD @@ -0,0 +1,684 @@ +../../../bin/pygmentize,sha256=DUtn7Ysdx0_KsM635berJ9OCX9Ip3ZMnbXLOukDH78M,251 +pygments-2.19.2.dist-info/INSTALLER,sha256=zuuue4knoyJ-UwPPXg8fezS7VCrXJQrAP7zeNuwvFQg,4 +pygments-2.19.2.dist-info/METADATA,sha256=euEA1n1nAGxkeYA92DX89HqbWfrHlEQeqOZqp_WYTYI,2512 +pygments-2.19.2.dist-info/RECORD,, +pygments-2.19.2.dist-info/WHEEL,sha256=qtCwoSJWgHk21S1Kb4ihdzI2rlJ1ZKaIurTj_ngOhyQ,87 +pygments-2.19.2.dist-info/entry_points.txt,sha256=uUXw-XhMKBEX4pWcCtpuTTnPhL3h7OEE2jWi51VQsa8,53 +pygments-2.19.2.dist-info/licenses/AUTHORS,sha256=BmDjGKbyFYAq3Icxq4XQxl_yfPzKP10oWX8wZHYZW9k,10824 +pygments-2.19.2.dist-info/licenses/LICENSE,sha256=qdZvHVJt8C4p3Oc0NtNOVuhjL0bCdbvf_HBWnogvnxc,1331 +pygments/__init__.py,sha256=_3UT86TGpHuW8FekdZ8uLidEZH1NhmcLiOy2KKNPCt4,2959 +pygments/__main__.py,sha256=p8AJyoyCOMYGvzWHdnq0_A9qaaVqaj02nIu3xhJp1_4,348 +pygments/__pycache__/__init__.cpython-312.pyc,, +pygments/__pycache__/__main__.cpython-312.pyc,, +pygments/__pycache__/cmdline.cpython-312.pyc,, +pygments/__pycache__/console.cpython-312.pyc,, +pygments/__pycache__/filter.cpython-312.pyc,, +pygments/__pycache__/formatter.cpython-312.pyc,, +pygments/__pycache__/lexer.cpython-312.pyc,, +pygments/__pycache__/modeline.cpython-312.pyc,, +pygments/__pycache__/plugin.cpython-312.pyc,, +pygments/__pycache__/regexopt.cpython-312.pyc,, +pygments/__pycache__/scanner.cpython-312.pyc,, +pygments/__pycache__/sphinxext.cpython-312.pyc,, +pygments/__pycache__/style.cpython-312.pyc,, +pygments/__pycache__/token.cpython-312.pyc,, +pygments/__pycache__/unistring.cpython-312.pyc,, +pygments/__pycache__/util.cpython-312.pyc,, +pygments/cmdline.py,sha256=4pL9Kpn2PUEKPobgrsQgg-vCx2NjsrapKzQ6LxQR7Q0,23536 +pygments/console.py,sha256=AagDWqwea2yBWf10KC9ptBgMpMjxKp8yABAmh-NQOVk,1718 +pygments/filter.py,sha256=YLtpTnZiu07nY3oK9nfR6E9Y1FBHhP5PX8gvkJWcfag,1910 +pygments/filters/__init__.py,sha256=B00KqPCQh5E0XhzaDK74Qa1E4fDSTlD6b0Pvr1v-vEQ,40344 +pygments/filters/__pycache__/__init__.cpython-312.pyc,, +pygments/formatter.py,sha256=H_4J-moKkKfRWUOW9J0u7hhw6n1LiO-2Xu1q2B0sE5w,4366 +pygments/formatters/__init__.py,sha256=7OuvmoYLyoPzoOQV_brHG8GSKYB_wjFSkAQng6x2y9g,5349 +pygments/formatters/__pycache__/__init__.cpython-312.pyc,, +pygments/formatters/__pycache__/_mapping.cpython-312.pyc,, +pygments/formatters/__pycache__/bbcode.cpython-312.pyc,, +pygments/formatters/__pycache__/groff.cpython-312.pyc,, +pygments/formatters/__pycache__/html.cpython-312.pyc,, +pygments/formatters/__pycache__/img.cpython-312.pyc,, +pygments/formatters/__pycache__/irc.cpython-312.pyc,, +pygments/formatters/__pycache__/latex.cpython-312.pyc,, +pygments/formatters/__pycache__/other.cpython-312.pyc,, +pygments/formatters/__pycache__/pangomarkup.cpython-312.pyc,, +pygments/formatters/__pycache__/rtf.cpython-312.pyc,, +pygments/formatters/__pycache__/svg.cpython-312.pyc,, +pygments/formatters/__pycache__/terminal.cpython-312.pyc,, +pygments/formatters/__pycache__/terminal256.cpython-312.pyc,, +pygments/formatters/_mapping.py,sha256=1Cw37FuQlNacnxRKmtlPX4nyLoX9_ttko5ZwscNUZZ4,4176 +pygments/formatters/bbcode.py,sha256=s0Ka35OKuIchoSgEAGf6rj0rl2a9ym9L31JVNSRbZFQ,3296 +pygments/formatters/groff.py,sha256=pLcIHj4jJS_lRAVFnyJODKDu1Xlyl9_AEIdOtbl3DT0,5082 +pygments/formatters/html.py,sha256=FrHJ69FUliEyPY0zTfab0C1gPf7LXsKgeRlhwkniqIs,35953 +pygments/formatters/img.py,sha256=aRpFo8mBmWTL3sBUjRCWkeS3rc6FZrSFC4EksDrl53g,23301 +pygments/formatters/irc.py,sha256=R0Js0TYWySlI2yE9sW6tN4d4X-x3k9ZmudsijGPnLmU,4945 +pygments/formatters/latex.py,sha256=BRYtbLeW_YD1kwhhnFInhJIKylurnri8CF1lP069KWE,19258 +pygments/formatters/other.py,sha256=8pYW27sU_7XicLUqOEt2yWSO0h1IEUM3TIv34KODLwo,4986 +pygments/formatters/pangomarkup.py,sha256=pcFvEC7K1Me0EjGeOZth4oCnEY85bfqc77XzZASEPpY,2206 +pygments/formatters/rtf.py,sha256=kcKMCxTXu-2-hpgEftlGJRm7Ss-yA_Sy8OsHH_qzykA,11921 +pygments/formatters/svg.py,sha256=R6A2ME6JsMQWFiyn8wcKwFUOD6vsu-HLwiIztLu-77E,7138 +pygments/formatters/terminal.py,sha256=J_F_dFXwR9LHWvatIDnwqRYJyjVmSo1Zx8K_XDh6SyM,4626 +pygments/formatters/terminal256.py,sha256=7GQFLE5cfmeu53CAzANO74-kBk2BFkXfn5phmZjYkhM,11717 +pygments/lexer.py,sha256=ib-F_0GxHkwGpb6vWP0DeLMLc7EYgjo3hWFKN5IgOq0,35109 +pygments/lexers/__init__.py,sha256=6YhzxGKlWk38P6JpIJUQ1rVvV0DEZjEmdYsdMQ58hSk,12067 +pygments/lexers/__pycache__/__init__.cpython-312.pyc,, +pygments/lexers/__pycache__/_ada_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_asy_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_cl_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_cocoa_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_csound_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_css_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_googlesql_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_julia_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_lasso_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_lilypond_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_lua_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_luau_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_mapping.cpython-312.pyc,, +pygments/lexers/__pycache__/_mql_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_mysql_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_openedge_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_php_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_postgres_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_qlik_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_scheme_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_scilab_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_sourcemod_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_sql_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_stan_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_stata_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_tsql_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_usd_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_vbscript_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/_vim_builtins.cpython-312.pyc,, +pygments/lexers/__pycache__/actionscript.cpython-312.pyc,, +pygments/lexers/__pycache__/ada.cpython-312.pyc,, +pygments/lexers/__pycache__/agile.cpython-312.pyc,, +pygments/lexers/__pycache__/algebra.cpython-312.pyc,, +pygments/lexers/__pycache__/ambient.cpython-312.pyc,, +pygments/lexers/__pycache__/amdgpu.cpython-312.pyc,, +pygments/lexers/__pycache__/ampl.cpython-312.pyc,, +pygments/lexers/__pycache__/apdlexer.cpython-312.pyc,, +pygments/lexers/__pycache__/apl.cpython-312.pyc,, +pygments/lexers/__pycache__/archetype.cpython-312.pyc,, +pygments/lexers/__pycache__/arrow.cpython-312.pyc,, +pygments/lexers/__pycache__/arturo.cpython-312.pyc,, +pygments/lexers/__pycache__/asc.cpython-312.pyc,, +pygments/lexers/__pycache__/asm.cpython-312.pyc,, +pygments/lexers/__pycache__/asn1.cpython-312.pyc,, +pygments/lexers/__pycache__/automation.cpython-312.pyc,, +pygments/lexers/__pycache__/bare.cpython-312.pyc,, +pygments/lexers/__pycache__/basic.cpython-312.pyc,, +pygments/lexers/__pycache__/bdd.cpython-312.pyc,, +pygments/lexers/__pycache__/berry.cpython-312.pyc,, +pygments/lexers/__pycache__/bibtex.cpython-312.pyc,, +pygments/lexers/__pycache__/blueprint.cpython-312.pyc,, +pygments/lexers/__pycache__/boa.cpython-312.pyc,, +pygments/lexers/__pycache__/bqn.cpython-312.pyc,, +pygments/lexers/__pycache__/business.cpython-312.pyc,, +pygments/lexers/__pycache__/c_cpp.cpython-312.pyc,, +pygments/lexers/__pycache__/c_like.cpython-312.pyc,, +pygments/lexers/__pycache__/capnproto.cpython-312.pyc,, +pygments/lexers/__pycache__/carbon.cpython-312.pyc,, +pygments/lexers/__pycache__/cddl.cpython-312.pyc,, +pygments/lexers/__pycache__/chapel.cpython-312.pyc,, +pygments/lexers/__pycache__/clean.cpython-312.pyc,, +pygments/lexers/__pycache__/codeql.cpython-312.pyc,, +pygments/lexers/__pycache__/comal.cpython-312.pyc,, +pygments/lexers/__pycache__/compiled.cpython-312.pyc,, +pygments/lexers/__pycache__/configs.cpython-312.pyc,, +pygments/lexers/__pycache__/console.cpython-312.pyc,, +pygments/lexers/__pycache__/cplint.cpython-312.pyc,, +pygments/lexers/__pycache__/crystal.cpython-312.pyc,, +pygments/lexers/__pycache__/csound.cpython-312.pyc,, +pygments/lexers/__pycache__/css.cpython-312.pyc,, +pygments/lexers/__pycache__/d.cpython-312.pyc,, +pygments/lexers/__pycache__/dalvik.cpython-312.pyc,, +pygments/lexers/__pycache__/data.cpython-312.pyc,, +pygments/lexers/__pycache__/dax.cpython-312.pyc,, +pygments/lexers/__pycache__/devicetree.cpython-312.pyc,, +pygments/lexers/__pycache__/diff.cpython-312.pyc,, +pygments/lexers/__pycache__/dns.cpython-312.pyc,, +pygments/lexers/__pycache__/dotnet.cpython-312.pyc,, +pygments/lexers/__pycache__/dsls.cpython-312.pyc,, +pygments/lexers/__pycache__/dylan.cpython-312.pyc,, +pygments/lexers/__pycache__/ecl.cpython-312.pyc,, +pygments/lexers/__pycache__/eiffel.cpython-312.pyc,, +pygments/lexers/__pycache__/elm.cpython-312.pyc,, +pygments/lexers/__pycache__/elpi.cpython-312.pyc,, +pygments/lexers/__pycache__/email.cpython-312.pyc,, +pygments/lexers/__pycache__/erlang.cpython-312.pyc,, +pygments/lexers/__pycache__/esoteric.cpython-312.pyc,, +pygments/lexers/__pycache__/ezhil.cpython-312.pyc,, +pygments/lexers/__pycache__/factor.cpython-312.pyc,, +pygments/lexers/__pycache__/fantom.cpython-312.pyc,, +pygments/lexers/__pycache__/felix.cpython-312.pyc,, +pygments/lexers/__pycache__/fift.cpython-312.pyc,, +pygments/lexers/__pycache__/floscript.cpython-312.pyc,, +pygments/lexers/__pycache__/forth.cpython-312.pyc,, +pygments/lexers/__pycache__/fortran.cpython-312.pyc,, +pygments/lexers/__pycache__/foxpro.cpython-312.pyc,, +pygments/lexers/__pycache__/freefem.cpython-312.pyc,, +pygments/lexers/__pycache__/func.cpython-312.pyc,, +pygments/lexers/__pycache__/functional.cpython-312.pyc,, +pygments/lexers/__pycache__/futhark.cpython-312.pyc,, +pygments/lexers/__pycache__/gcodelexer.cpython-312.pyc,, +pygments/lexers/__pycache__/gdscript.cpython-312.pyc,, +pygments/lexers/__pycache__/gleam.cpython-312.pyc,, +pygments/lexers/__pycache__/go.cpython-312.pyc,, +pygments/lexers/__pycache__/grammar_notation.cpython-312.pyc,, +pygments/lexers/__pycache__/graph.cpython-312.pyc,, +pygments/lexers/__pycache__/graphics.cpython-312.pyc,, +pygments/lexers/__pycache__/graphql.cpython-312.pyc,, +pygments/lexers/__pycache__/graphviz.cpython-312.pyc,, +pygments/lexers/__pycache__/gsql.cpython-312.pyc,, +pygments/lexers/__pycache__/hare.cpython-312.pyc,, +pygments/lexers/__pycache__/haskell.cpython-312.pyc,, +pygments/lexers/__pycache__/haxe.cpython-312.pyc,, +pygments/lexers/__pycache__/hdl.cpython-312.pyc,, +pygments/lexers/__pycache__/hexdump.cpython-312.pyc,, +pygments/lexers/__pycache__/html.cpython-312.pyc,, +pygments/lexers/__pycache__/idl.cpython-312.pyc,, +pygments/lexers/__pycache__/igor.cpython-312.pyc,, +pygments/lexers/__pycache__/inferno.cpython-312.pyc,, +pygments/lexers/__pycache__/installers.cpython-312.pyc,, +pygments/lexers/__pycache__/int_fiction.cpython-312.pyc,, +pygments/lexers/__pycache__/iolang.cpython-312.pyc,, +pygments/lexers/__pycache__/j.cpython-312.pyc,, +pygments/lexers/__pycache__/javascript.cpython-312.pyc,, +pygments/lexers/__pycache__/jmespath.cpython-312.pyc,, +pygments/lexers/__pycache__/jslt.cpython-312.pyc,, +pygments/lexers/__pycache__/json5.cpython-312.pyc,, +pygments/lexers/__pycache__/jsonnet.cpython-312.pyc,, +pygments/lexers/__pycache__/jsx.cpython-312.pyc,, +pygments/lexers/__pycache__/julia.cpython-312.pyc,, +pygments/lexers/__pycache__/jvm.cpython-312.pyc,, +pygments/lexers/__pycache__/kuin.cpython-312.pyc,, +pygments/lexers/__pycache__/kusto.cpython-312.pyc,, +pygments/lexers/__pycache__/ldap.cpython-312.pyc,, +pygments/lexers/__pycache__/lean.cpython-312.pyc,, +pygments/lexers/__pycache__/lilypond.cpython-312.pyc,, +pygments/lexers/__pycache__/lisp.cpython-312.pyc,, +pygments/lexers/__pycache__/macaulay2.cpython-312.pyc,, +pygments/lexers/__pycache__/make.cpython-312.pyc,, +pygments/lexers/__pycache__/maple.cpython-312.pyc,, +pygments/lexers/__pycache__/markup.cpython-312.pyc,, +pygments/lexers/__pycache__/math.cpython-312.pyc,, +pygments/lexers/__pycache__/matlab.cpython-312.pyc,, +pygments/lexers/__pycache__/maxima.cpython-312.pyc,, +pygments/lexers/__pycache__/meson.cpython-312.pyc,, +pygments/lexers/__pycache__/mime.cpython-312.pyc,, +pygments/lexers/__pycache__/minecraft.cpython-312.pyc,, +pygments/lexers/__pycache__/mips.cpython-312.pyc,, +pygments/lexers/__pycache__/ml.cpython-312.pyc,, +pygments/lexers/__pycache__/modeling.cpython-312.pyc,, +pygments/lexers/__pycache__/modula2.cpython-312.pyc,, +pygments/lexers/__pycache__/mojo.cpython-312.pyc,, +pygments/lexers/__pycache__/monte.cpython-312.pyc,, +pygments/lexers/__pycache__/mosel.cpython-312.pyc,, +pygments/lexers/__pycache__/ncl.cpython-312.pyc,, +pygments/lexers/__pycache__/nimrod.cpython-312.pyc,, +pygments/lexers/__pycache__/nit.cpython-312.pyc,, +pygments/lexers/__pycache__/nix.cpython-312.pyc,, +pygments/lexers/__pycache__/numbair.cpython-312.pyc,, +pygments/lexers/__pycache__/oberon.cpython-312.pyc,, +pygments/lexers/__pycache__/objective.cpython-312.pyc,, +pygments/lexers/__pycache__/ooc.cpython-312.pyc,, +pygments/lexers/__pycache__/openscad.cpython-312.pyc,, +pygments/lexers/__pycache__/other.cpython-312.pyc,, +pygments/lexers/__pycache__/parasail.cpython-312.pyc,, +pygments/lexers/__pycache__/parsers.cpython-312.pyc,, +pygments/lexers/__pycache__/pascal.cpython-312.pyc,, +pygments/lexers/__pycache__/pawn.cpython-312.pyc,, +pygments/lexers/__pycache__/pddl.cpython-312.pyc,, +pygments/lexers/__pycache__/perl.cpython-312.pyc,, +pygments/lexers/__pycache__/phix.cpython-312.pyc,, +pygments/lexers/__pycache__/php.cpython-312.pyc,, +pygments/lexers/__pycache__/pointless.cpython-312.pyc,, +pygments/lexers/__pycache__/pony.cpython-312.pyc,, +pygments/lexers/__pycache__/praat.cpython-312.pyc,, +pygments/lexers/__pycache__/procfile.cpython-312.pyc,, +pygments/lexers/__pycache__/prolog.cpython-312.pyc,, +pygments/lexers/__pycache__/promql.cpython-312.pyc,, +pygments/lexers/__pycache__/prql.cpython-312.pyc,, +pygments/lexers/__pycache__/ptx.cpython-312.pyc,, +pygments/lexers/__pycache__/python.cpython-312.pyc,, +pygments/lexers/__pycache__/q.cpython-312.pyc,, +pygments/lexers/__pycache__/qlik.cpython-312.pyc,, +pygments/lexers/__pycache__/qvt.cpython-312.pyc,, +pygments/lexers/__pycache__/r.cpython-312.pyc,, +pygments/lexers/__pycache__/rdf.cpython-312.pyc,, +pygments/lexers/__pycache__/rebol.cpython-312.pyc,, +pygments/lexers/__pycache__/rego.cpython-312.pyc,, +pygments/lexers/__pycache__/resource.cpython-312.pyc,, +pygments/lexers/__pycache__/ride.cpython-312.pyc,, +pygments/lexers/__pycache__/rita.cpython-312.pyc,, +pygments/lexers/__pycache__/rnc.cpython-312.pyc,, +pygments/lexers/__pycache__/roboconf.cpython-312.pyc,, +pygments/lexers/__pycache__/robotframework.cpython-312.pyc,, +pygments/lexers/__pycache__/ruby.cpython-312.pyc,, +pygments/lexers/__pycache__/rust.cpython-312.pyc,, +pygments/lexers/__pycache__/sas.cpython-312.pyc,, +pygments/lexers/__pycache__/savi.cpython-312.pyc,, +pygments/lexers/__pycache__/scdoc.cpython-312.pyc,, +pygments/lexers/__pycache__/scripting.cpython-312.pyc,, +pygments/lexers/__pycache__/sgf.cpython-312.pyc,, +pygments/lexers/__pycache__/shell.cpython-312.pyc,, +pygments/lexers/__pycache__/sieve.cpython-312.pyc,, +pygments/lexers/__pycache__/slash.cpython-312.pyc,, +pygments/lexers/__pycache__/smalltalk.cpython-312.pyc,, +pygments/lexers/__pycache__/smithy.cpython-312.pyc,, +pygments/lexers/__pycache__/smv.cpython-312.pyc,, +pygments/lexers/__pycache__/snobol.cpython-312.pyc,, +pygments/lexers/__pycache__/solidity.cpython-312.pyc,, +pygments/lexers/__pycache__/soong.cpython-312.pyc,, +pygments/lexers/__pycache__/sophia.cpython-312.pyc,, +pygments/lexers/__pycache__/special.cpython-312.pyc,, +pygments/lexers/__pycache__/spice.cpython-312.pyc,, +pygments/lexers/__pycache__/sql.cpython-312.pyc,, +pygments/lexers/__pycache__/srcinfo.cpython-312.pyc,, +pygments/lexers/__pycache__/stata.cpython-312.pyc,, +pygments/lexers/__pycache__/supercollider.cpython-312.pyc,, +pygments/lexers/__pycache__/tablegen.cpython-312.pyc,, +pygments/lexers/__pycache__/tact.cpython-312.pyc,, +pygments/lexers/__pycache__/tal.cpython-312.pyc,, +pygments/lexers/__pycache__/tcl.cpython-312.pyc,, +pygments/lexers/__pycache__/teal.cpython-312.pyc,, +pygments/lexers/__pycache__/templates.cpython-312.pyc,, +pygments/lexers/__pycache__/teraterm.cpython-312.pyc,, +pygments/lexers/__pycache__/testing.cpython-312.pyc,, +pygments/lexers/__pycache__/text.cpython-312.pyc,, +pygments/lexers/__pycache__/textedit.cpython-312.pyc,, +pygments/lexers/__pycache__/textfmts.cpython-312.pyc,, +pygments/lexers/__pycache__/theorem.cpython-312.pyc,, +pygments/lexers/__pycache__/thingsdb.cpython-312.pyc,, +pygments/lexers/__pycache__/tlb.cpython-312.pyc,, +pygments/lexers/__pycache__/tls.cpython-312.pyc,, +pygments/lexers/__pycache__/tnt.cpython-312.pyc,, +pygments/lexers/__pycache__/trafficscript.cpython-312.pyc,, +pygments/lexers/__pycache__/typoscript.cpython-312.pyc,, +pygments/lexers/__pycache__/typst.cpython-312.pyc,, +pygments/lexers/__pycache__/ul4.cpython-312.pyc,, +pygments/lexers/__pycache__/unicon.cpython-312.pyc,, +pygments/lexers/__pycache__/urbi.cpython-312.pyc,, +pygments/lexers/__pycache__/usd.cpython-312.pyc,, +pygments/lexers/__pycache__/varnish.cpython-312.pyc,, +pygments/lexers/__pycache__/verification.cpython-312.pyc,, +pygments/lexers/__pycache__/verifpal.cpython-312.pyc,, +pygments/lexers/__pycache__/vip.cpython-312.pyc,, +pygments/lexers/__pycache__/vyper.cpython-312.pyc,, +pygments/lexers/__pycache__/web.cpython-312.pyc,, +pygments/lexers/__pycache__/webassembly.cpython-312.pyc,, +pygments/lexers/__pycache__/webidl.cpython-312.pyc,, +pygments/lexers/__pycache__/webmisc.cpython-312.pyc,, +pygments/lexers/__pycache__/wgsl.cpython-312.pyc,, +pygments/lexers/__pycache__/whiley.cpython-312.pyc,, +pygments/lexers/__pycache__/wowtoc.cpython-312.pyc,, +pygments/lexers/__pycache__/wren.cpython-312.pyc,, +pygments/lexers/__pycache__/x10.cpython-312.pyc,, +pygments/lexers/__pycache__/xorg.cpython-312.pyc,, +pygments/lexers/__pycache__/yang.cpython-312.pyc,, +pygments/lexers/__pycache__/yara.cpython-312.pyc,, +pygments/lexers/__pycache__/zig.cpython-312.pyc,, +pygments/lexers/_ada_builtins.py,sha256=CA_OnShtdc7wWh9oYcRlcrkDAQwYUKl6w7tdSbALQd4,1543 +pygments/lexers/_asy_builtins.py,sha256=cd9M00YH19w5ZL7aqucmC3nwpJGTS04U-01NLy5E2_4,27287 +pygments/lexers/_cl_builtins.py,sha256=kQeUIyZjP4kX0frkICDcKxBYQCLqzIDXa5WV5cevhDo,13994 +pygments/lexers/_cocoa_builtins.py,sha256=Ka1lLJe7JfWtdho4IFIB82X9yBvrbfHCCmEG-peXXhQ,105173 +pygments/lexers/_csound_builtins.py,sha256=qnQYKeI26ZHim316uqy_hDiRiCoHo2RHjD3sYBALyXs,18414 +pygments/lexers/_css_builtins.py,sha256=aD-dhLFXVd1Atn_bZd7gEdQn7Mhe60_VHpvZ340WzDI,12446 +pygments/lexers/_googlesql_builtins.py,sha256=IkrOk-T2v1yzbGzUEEQh5_Cf4uC_cmL_uuhwDpZlTug,16132 +pygments/lexers/_julia_builtins.py,sha256=N2WdSw5zgI2fhDat_i4YeVqurRTC_P8x71ez00SCN6U,11883 +pygments/lexers/_lasso_builtins.py,sha256=8q1gbsrMJeaeUhxIYKhaOxC9j_B-NBpq_XFj2Ze41X0,134510 +pygments/lexers/_lilypond_builtins.py,sha256=XTbGL1z1oKMoqWLEktG33jx5GdGTI9CpeO5NheEi4Y0,108094 +pygments/lexers/_lua_builtins.py,sha256=PhFdZV5-Tzz2j_q4lvG9lr84ELGfL41BhnrSDNNTaG4,8108 +pygments/lexers/_luau_builtins.py,sha256=-IDrU04kUVfjXwSQzMMpXmMYhNsQxZVVZk8cuAA0Lo0,955 +pygments/lexers/_mapping.py,sha256=9fv7xYOUAOr6LzfdFS4MDbPu78o4OQQH-2nsI1bNZf4,70438 +pygments/lexers/_mql_builtins.py,sha256=ybRQjlb7Cul0sDstnzxJl3h0qS6Ieqsr811fqrxyumU,24713 +pygments/lexers/_mysql_builtins.py,sha256=y0kAWZVAs0z2dTFJJV42OZpILgRnd8T3zSlBFv-g_oA,25838 +pygments/lexers/_openedge_builtins.py,sha256=Sz4j9-CPWIaxMa-2fZgY66j7igcu1ob1GR2UtI8zAkg,49398 +pygments/lexers/_php_builtins.py,sha256=Jd4BZpjMDELPi4EVoSxK1-8BFTc63HUwYfm1rLrGj0M,107922 +pygments/lexers/_postgres_builtins.py,sha256=Pqh4z0RBRbnW6rCQtWUdzWCJxNyqpJ7_0HOktxHDxk4,13343 +pygments/lexers/_qlik_builtins.py,sha256=xuJy9c9uZDXv6h8z582P5PrxqkxTZ_nS8gPl9OD9VN8,12595 +pygments/lexers/_scheme_builtins.py,sha256=2hNtJOJmP21lUsikpqMJ2gAmLT3Rwn_KEeqhXwCjgfk,32564 +pygments/lexers/_scilab_builtins.py,sha256=oZYPB1XPdIEz3pII11pFDe6extRRyWGA7pY06X8KZ8w,52411 +pygments/lexers/_sourcemod_builtins.py,sha256=H8AFLsNDdEpymIWOpDwbDJGCP1w-x-1gSlzPDioMF4o,26777 +pygments/lexers/_sql_builtins.py,sha256=oe8F9wWuO2iS6nEsZAdJtCUChBTjgM1Sq_aipu74jXM,6767 +pygments/lexers/_stan_builtins.py,sha256=dwi1hllM_NsaCv-aXJy7lEi57X5Hh5gSD97aCQyT9KM,13445 +pygments/lexers/_stata_builtins.py,sha256=Hqrr6j77zWU3cGGpBPohwexZci43YA4_sVYE4E1sNow,27227 +pygments/lexers/_tsql_builtins.py,sha256=Pi2RhTXcLE3glI9oxNhyVsOMn-fK_1TRxJ-EsYP5LcI,15460 +pygments/lexers/_usd_builtins.py,sha256=c9hbU1cwqBUCFIhNfu_Dob8ywv1rlPhi9w2OTj3kR8s,1658 +pygments/lexers/_vbscript_builtins.py,sha256=MqJ2ABywD21aSRtWYZRG64CCbGstC1kfsiHGJmZzxiw,4225 +pygments/lexers/_vim_builtins.py,sha256=bA4mH8t1mPPQfEiUCKEqRO1O0rL2DUG0Ux1Bt8ZSu0E,57066 +pygments/lexers/actionscript.py,sha256=JBngCe5UhYT_0dLD2j7PnPO0xRRJhmypEuQ-C5in8pY,11727 +pygments/lexers/ada.py,sha256=58k5ra1vGS4iLpW3h1ItY9ftzF3WevaeAAXzAYTiYkQ,5353 +pygments/lexers/agile.py,sha256=DN-7AVIqtG1MshA94rtSGYI_884hVHgzq405wD0_dl8,896 +pygments/lexers/algebra.py,sha256=yGTu9Tt-cQzAISQYIC5MS5a3z4QmL-tGcXnd_pkWGbk,9952 +pygments/lexers/ambient.py,sha256=UnzKpIlfSm3iitHvMd7XTMSY8TjZYYhKOC3AiARS_cE,2605 +pygments/lexers/amdgpu.py,sha256=S8qjn2UMLhBFm3Yn_c06XAGf8cl5x_ZeluelWG_-JAw,1723 +pygments/lexers/ampl.py,sha256=ZBRfDXm760gR1a1gqItnsHuoO3JdUcTBjJ5tFY9UtPA,4176 +pygments/lexers/apdlexer.py,sha256=Zr5-jgjxC8PKzRlEeclakZXPHci7FHBZghQ6wwiuT7A,30800 +pygments/lexers/apl.py,sha256=PTQMp-bxT5P-DbrEvFha10HBTcsDJ5srL3I1s9ljz58,3404 +pygments/lexers/archetype.py,sha256=pQVlP1Fb5OA8nn7QwmFaaaOSvvpoIsQVw43FVCQCve4,11538 +pygments/lexers/arrow.py,sha256=2PKdbWq3xQLF1KoDbWvSxpjwKRrznnDiArTflRGZzBo,3564 +pygments/lexers/arturo.py,sha256=U5MtRNHJtnBn4ZOeWmW6MKlVRG7SX6KhTRamDqzn9tA,11414 +pygments/lexers/asc.py,sha256=-DgZl9jccBDHPlDmjCsrEqx0-Q7ap7XVdNKtxLNWG1w,1693 +pygments/lexers/asm.py,sha256=xm2Y5mcT-sF3oQvair4SWs9EWTyndoaUoSsDy5v6shI,41967 +pygments/lexers/asn1.py,sha256=BlcloIX2bu6Q7BxGcksuhYFHGsXLVKyB4B9mFd4Pj6E,4262 +pygments/lexers/automation.py,sha256=Q61qon8EwpfakMh_2MS2E2zUUT16rG3UNIKPYjITeTs,19831 +pygments/lexers/bare.py,sha256=tWoei86JJX1k-ADhaXd5TgX6ItDTici9yFWpkTPhnfM,3020 +pygments/lexers/basic.py,sha256=qpVe5h8Fa7NJo1EihN-4R_UZpHO6my2Ssgkb-BktkKs,27989 +pygments/lexers/bdd.py,sha256=yysefcOFAEyk9kJ2y4EXmzJTecgLYUHlWixt_3YzPMU,1641 +pygments/lexers/berry.py,sha256=zxGowFb8HMIyN15-m8nmWnW6bPRR4esKtSEVugc9uXM,3209 +pygments/lexers/bibtex.py,sha256=yuNoPxwrJf9DCGUT17hxfDzbq_HtCLkQkRbBtiTVmeQ,4811 +pygments/lexers/blueprint.py,sha256=NzvWHMxCLDWt8hc6gB5jokltxVJgNa7Jwh4c61ng388,6188 +pygments/lexers/boa.py,sha256=dOot1XWNZThPIio2UyAX67K6EpISjSRCFjotD7dcnwE,3921 +pygments/lexers/bqn.py,sha256=nJiwrPKKbRF-qdai5tfqipwBkkko2P3weiZAjHUMimY,3671 +pygments/lexers/business.py,sha256=lRtekOJfsDkb12AGbuz10-G67OJrVJgCBtihTQ8_aoY,28345 +pygments/lexers/c_cpp.py,sha256=D7ZIswaHASlGBgoTlwnSqTQHf8_JyvvSt2L2q1W-F6g,18059 +pygments/lexers/c_like.py,sha256=FTGp17ds6X2rDZOHup2hH6BEn3gKK4nLm9pydNEhm0E,32021 +pygments/lexers/capnproto.py,sha256=XQJAh1WS-0ulqbTn9TdzR6gEgWLcuBqb4sj3jNsrhsY,2174 +pygments/lexers/carbon.py,sha256=av12YuTGZGpOa1Cmxp3lppx3LfSJUWbvOu0ixmUVll0,3211 +pygments/lexers/cddl.py,sha256=MKa70IwABgjBjYu15_Q9v8rsu2sr1a-i2jkiaPTI6sM,5076 +pygments/lexers/chapel.py,sha256=0n_fL3ehLC4pw4YKnmq9jxIXOJcxGPka1Wr1t1zsXPc,5156 +pygments/lexers/clean.py,sha256=dkDPAwF5BTALPeuKFoRKOSD3RfsKcGWbaRo6_G8LHng,6418 +pygments/lexers/codeql.py,sha256=ebvghn2zbrnETV4buVozMDmRCVKSdGiIN8ycLlHpGsE,2576 +pygments/lexers/comal.py,sha256=TC3NzcJ58ew5jw7qwK0kJ-okTA47psZje0yAIS39HR4,3179 +pygments/lexers/compiled.py,sha256=Slfo1sjWqcPawUwf0dIIZLBCL5pkOIoAX2S8Lxs02Mc,1426 +pygments/lexers/configs.py,sha256=wW8pY0Sa5a10pnAeTLGf48HhixQTVageIyHEf1aYMCc,50913 +pygments/lexers/console.py,sha256=-jAG120dupvV3kG3zC70brLJvSLwTFqMubBQuj_GVnU,4180 +pygments/lexers/cplint.py,sha256=DkbyE5EKydLgf6BRr1FhQrK-IeQPL7Zmjk0DVdlRFnQ,1389 +pygments/lexers/crystal.py,sha256=xU-RnpIkpjrquoxtOuOcP8fcesSJl4xhU7kO9m42LZY,15754 +pygments/lexers/csound.py,sha256=ioSw4Q04wdwjUAbnTZ1qLhUq1vxdWFxhh3QtEl5RAJc,16998 +pygments/lexers/css.py,sha256=JN1RBYsee-jrpHWrSmhN3TKc4TkOBn-_BEGpgTCzcqE,25376 +pygments/lexers/d.py,sha256=piOy0EJeiAwPHugiM3gVv0z7HNh3u2gZQoCUSASRbY4,9920 +pygments/lexers/dalvik.py,sha256=deFg2JPBktJ9mEGb9EgxNkmd6vaMjJFQVzUHo8NKIa8,4606 +pygments/lexers/data.py,sha256=o0x0SmB5ms_CPUPljEEEenOON4IQWn86DkwFjkJYCOg,27026 +pygments/lexers/dax.py,sha256=ASi73qmr7OA7cVZXF2GTYGt01Ly1vY8CgD_Pnpm8k-4,8098 +pygments/lexers/devicetree.py,sha256=RecSQCidt8DRE1QFCPUbwwR0hiRlNtsFihdGldeUn3k,4019 +pygments/lexers/diff.py,sha256=F6vxZ64wm5Nag_97de1H_3F700ZwCVnYjKvtT5jilww,5382 +pygments/lexers/dns.py,sha256=Hh5hJ7MXfrq36KgfyIRwK3X8o1LdR98IKERcV4eZ7HY,3891 +pygments/lexers/dotnet.py,sha256=NDE0kOmpe96GLO-zwNLazmj77E9ORGmKpa4ZMCXDXxQ,39441 +pygments/lexers/dsls.py,sha256=GnHKhGL5GxsRFnqC7-65NTPZLOZdmnllNrGP86x_fQE,36746 +pygments/lexers/dylan.py,sha256=7zZ1EbHWXeVHqTD36AqykKqo3fhuIh4sM-whcxUaH_Y,10409 +pygments/lexers/ecl.py,sha256=vhmpa2LBrHxsPkYcf3kPZ1ItVaLRDTebi186wY0xGZA,6371 +pygments/lexers/eiffel.py,sha256=5ydYIEFcgcMoEj4BlK31hZ0aJb8OX0RdAvuCNdlxwqw,2690 +pygments/lexers/elm.py,sha256=uRCddU8jK5vVkH6Y66y8KOsDJprIfrOgeYq3hv1PxAM,3152 +pygments/lexers/elpi.py,sha256=O9j_WKBPyvNFjCRuPciVpW4etVSnILm_T79BhCPZYmo,6877 +pygments/lexers/email.py,sha256=ZZL6yvwCRl1CEQyysuOu0lbabp5tjMutS7f3efFKGR4,4804 +pygments/lexers/erlang.py,sha256=bU11eVHvooLwmVknzN6Xkb2DMk7HbenqdNlYSzhThDM,19147 +pygments/lexers/esoteric.py,sha256=Jfp8UUKyKYsqLaqXRZT3GSM9dzkF65zduwfnH1GoGhU,10500 +pygments/lexers/ezhil.py,sha256=22r-xjvvBVpExTqCI-HycAwunDb1p5gY4tIfDmM0vDw,3272 +pygments/lexers/factor.py,sha256=urZ4En4uKFCLXdEkXLWg9EYUFGHQTTDCwNXtyq-ngok,19530 +pygments/lexers/fantom.py,sha256=JJ13-NwykD-iIESnuzCefCYeQDO95cHMJA8TasF4gHA,10231 +pygments/lexers/felix.py,sha256=F-v0si4zPtRelqzDQWXI1-tarCE-BvawziODxRU7378,9655 +pygments/lexers/fift.py,sha256=rOCwp3v5ocK5YOWvt7Td3Md--97_8e-7Sonx52uS8mA,1644 +pygments/lexers/floscript.py,sha256=aHh82k52jMuDuzl9LatrcSANJiXTCyjGU3SO53bwbb0,2667 +pygments/lexers/forth.py,sha256=ZMtsHdNbnS_0IdSYlfAlfTSPEr0MEsRo-YZriQNueTQ,7193 +pygments/lexers/fortran.py,sha256=1PE5dTxf4Df6LUeXFcmNtyeXWsC8tSiK5dYwPHIJeeQ,10382 +pygments/lexers/foxpro.py,sha256=CBkW62Fuibz3yfyelZCaEO8GGdFJWsuRhqwtsSeBwLM,26295 +pygments/lexers/freefem.py,sha256=LFBQk-m1-nNCgrl-VDH3QwnVWurvb7W29i06LoT207A,26913 +pygments/lexers/func.py,sha256=OR2rkM7gf9fKvad5WcFQln-_U_pb-RUCM9eQatToF4A,3700 +pygments/lexers/functional.py,sha256=fYT2AGZ642cRkIAId0rnXFBsx1c8LLEDRN_VuCEkUyM,693 +pygments/lexers/futhark.py,sha256=Vf1i4t-tR3zqaktVjhTzFNg_ts_9CcyA4ZDfDizbCmk,3743 +pygments/lexers/gcodelexer.py,sha256=4Xs9ax4-JZGupW_qSnHon39wQGpb-tNA3xorMKg841E,874 +pygments/lexers/gdscript.py,sha256=Ws7JKxy0M0IyZ_1iMfRvJPrizEwmeCNLDoeMIFaM-CU,7566 +pygments/lexers/gleam.py,sha256=XIlTcq6cB743pCqbNYo8PocSkjZyDPR6hHgdaJNJ1Vc,2392 +pygments/lexers/go.py,sha256=4LezefgyuqZWHzLZHieUkKTi-ssY6aHJxx7Z-LFaLK0,3783 +pygments/lexers/grammar_notation.py,sha256=LvzhRQHgwZzq9oceukZS_hwnKK58ee7Z5d0cwXOR734,8043 +pygments/lexers/graph.py,sha256=WFqoPA1c_hHYrV0i_F7-eUw3Co4_HmZY3GJ-TyDr670,4108 +pygments/lexers/graphics.py,sha256=tmF9NNALnvPnax8ywYC3pLOla45YXtp9UA0H-5EiTQY,39145 +pygments/lexers/graphql.py,sha256=O_zcrGrBaDaKTlUoJGRruxqk7CJi-NR92Y0Cs-KkCvw,5601 +pygments/lexers/graphviz.py,sha256=mzdXOMpwz9_V-be1eTAMyhkKCBl6UxCIXuq6C2yrtsw,1934 +pygments/lexers/gsql.py,sha256=VPZk9sb26-DumRkWfEaSTeoc0lx5xt5n-6eDDLezMtc,3990 +pygments/lexers/hare.py,sha256=PGCOuILktJsmtTpCZZKkMFtObfJuBpei8HM8HHuq1Tw,2649 +pygments/lexers/haskell.py,sha256=MYr74-PAC8kGJRX-dZmvZsHTc7a2u6yFS2B19LfDD7g,33262 +pygments/lexers/haxe.py,sha256=WHCy_nrXHnfLITfbdp3Ji3lqQU4HAsTUpXsLCp2_4sk,30974 +pygments/lexers/hdl.py,sha256=MOWxhmAuE4Ei0CKDqqaON7T8tl43geancrNYM136Z0U,22738 +pygments/lexers/hexdump.py,sha256=1lj9oJ-KiZXSVYvTMfGmEAQzNEW08WlMcC2I5aYvHK4,3653 +pygments/lexers/html.py,sha256=MxYTI4EeT7QxoGleCAyQq-8n_Sgly6tD95H5zanCNmk,21977 +pygments/lexers/idl.py,sha256=rcihUAGhfuGEaSW6pgFq6NzplT_pv0DagUoefg4zAmk,15449 +pygments/lexers/igor.py,sha256=wVefbUjb3ftaW3LCKGtX1JgLgiY4EmRor5gVOn8vQA8,31633 +pygments/lexers/inferno.py,sha256=ChE_5y5SLH_75Uv7D2dKWQMk2dlN6z1gY1IDjlJZ8rU,3135 +pygments/lexers/installers.py,sha256=ZHliit4Pxz1tYKOIjKkDXI5djTkpzYUMVIPR1xvUrL8,14435 +pygments/lexers/int_fiction.py,sha256=0ZzIa1sZDUQsltd1oHuS-BoNiOF8zKQfcVuDyK1Ttv8,56544 +pygments/lexers/iolang.py,sha256=L6dNDCLH0kxkIUi00fI4Z14QnRu79UcNDrgv02c5Zw8,1905 +pygments/lexers/j.py,sha256=DqNdwQGFLiZW3mCNLRg81gpmsy4Hgcai_9NP3LbWhNU,4853 +pygments/lexers/javascript.py,sha256=TGKQLSrCprCKfhLLGAq_0EOdvqvJKX9pOdKo7tCRurQ,63243 +pygments/lexers/jmespath.py,sha256=R5yA5LJ2nTIaDwnFIpSNGAThd0sAYFccwawA9xBptlg,2082 +pygments/lexers/jslt.py,sha256=OeYQf8O2_9FCaf9W6Q3a7rPdAFLthePCtVSgCrOTcl8,3700 +pygments/lexers/json5.py,sha256=8JZbc8EiTEZdKaIdQg3hXEh0mHWSzPlwd473a0nUuT0,2502 +pygments/lexers/jsonnet.py,sha256=bx2G6J4tJqGrJV1PyZrIWzWHXcoefCX-4lIxxtbn2gw,5636 +pygments/lexers/jsx.py,sha256=wGsoGSB40qAJrVfXwRPtan7OcK0O87RVsHHk0m6gogk,2693 +pygments/lexers/julia.py,sha256=0ZDJ9X83V5GqJzA6T6p0TTN8WHy2JAjvu-FSBXvfXdc,11710 +pygments/lexers/jvm.py,sha256=Yt1iQ3QodXRY-x_HUOGedhyuBBHn5jYH-I8NzOzHTlE,72667 +pygments/lexers/kuin.py,sha256=3dKKJVJlskgrvMKv2tY9NOsFfDjyo-3MLcJ1lFKdXSg,11405 +pygments/lexers/kusto.py,sha256=kaxkoPpEBDsBTCvCOkZZx7oGfv0jk_UNIRIRbfVAsBE,3477 +pygments/lexers/ldap.py,sha256=77vF4t_19x9V522cxRCM5d3HW8Ne3giYsFsMPVYYBw4,6551 +pygments/lexers/lean.py,sha256=7HWRgxFsxS1N9XKqw0vfKwaxl27s5YiVYtZeRUoTHFo,8570 +pygments/lexers/lilypond.py,sha256=yd2Tuv67um6EyCIr-VwBnlPhTHxMaQsBJ4nGgO5fjIk,9752 +pygments/lexers/lisp.py,sha256=EHUy1g4pzEsYPE-zGj2rAXm3YATE1j9dCQOr5-JPSkU,157668 +pygments/lexers/macaulay2.py,sha256=zkV-vxjQYa0Jj9TGfFP1iMgpTZ4ApQuAAIdJVGWb2is,33366 +pygments/lexers/make.py,sha256=YMI5DBCrxWca-pz9cVXcyfuHLcikPx9R_3pW_98Myqo,7831 +pygments/lexers/maple.py,sha256=Rs0dEmOMD3C1YQPd0mntN-vzReq4XfHegH6xV4lvJWo,7960 +pygments/lexers/markup.py,sha256=zWtxsyIx_1OxQzS6wLe8bEqglePv4RqvJjbia8AvV5c,65088 +pygments/lexers/math.py,sha256=P3ZK1ePd8ZnLdlmHezo2irCA8T2-nlHBoSaBoT5mEVI,695 +pygments/lexers/matlab.py,sha256=F9KO4qowIhfP8oVhCRRzE_1sqg4zmQbsB2NZH193PiM,133027 +pygments/lexers/maxima.py,sha256=a0h9Ggs9JEovTrzbJT-BLVbOqI29yPnaMZlkU5f_FeY,2715 +pygments/lexers/meson.py,sha256=BMrsDo6BH2lzTFw7JDwQ9SDNMTrRkXCNRDVf4aFHdsI,4336 +pygments/lexers/mime.py,sha256=yGrf3h37LK4b6ERBpFiL_qzn3JgOfGR5KLagnbWFl6c,7582 +pygments/lexers/minecraft.py,sha256=Nu88snDDPzM0D-742fFdUriczL-EE911pAd4_I4-pAw,13696 +pygments/lexers/mips.py,sha256=STKiZT67b3QERXXn7XKVxlPBu7vwbPC5EyCpuf3Jfbw,4656 +pygments/lexers/ml.py,sha256=t8sCv4BjvuBq6AihKKUwStEONIgdXCC2RMtO0RopNbM,35390 +pygments/lexers/modeling.py,sha256=M7B58bGB-Zwd1EmPxKqtRvg7TgNCyem3MVUHv0_H2SQ,13683 +pygments/lexers/modula2.py,sha256=NtpXBRoUCeHfflgB39LknSkCwhBHBKv2Er_pinjVsNE,53072 +pygments/lexers/mojo.py,sha256=8JRVoftN1E-W2woG0K-4n8PQXTUM9iY6Sl5sWb2uGNg,24233 +pygments/lexers/monte.py,sha256=baWU6zlXloenw9MO1MtEVGE9i3CfiXAYhqU621MIjRk,6289 +pygments/lexers/mosel.py,sha256=gjRdedhA1jTjoYoM1Gpaoog_I9o7TRbYMHk97N1TXwg,9297 +pygments/lexers/ncl.py,sha256=zJ6ahlitit4S0pBXc7Wu96PB7xOn59MwfR2HdY5_C60,63999 +pygments/lexers/nimrod.py,sha256=Q1NSqEkLC5wWt7xJyKC-vzWw_Iw2SfDNP_pyMFBuIfA,6413 +pygments/lexers/nit.py,sha256=p_hVD8GzMRl3CABVKHtYgnXFUQk0i5F2FbWFA6WXm6s,2725 +pygments/lexers/nix.py,sha256=NOrv20gdq-2A7eZ6c2gElPHv1Xx2pvv20-qOymL9GMg,4421 +pygments/lexers/numbair.py,sha256=fxkp2CXeXWKBMewfi1H4JSYkmm4kU58wZ2Sh9BDYAWQ,1758 +pygments/lexers/oberon.py,sha256=jw403qUUs7zpTHAs5CbLjb8qiuwtxLk0spDIYqGZwAw,4210 +pygments/lexers/objective.py,sha256=Fo1WB3JMj8sNeYnvB84H4_qwhOt4WNJtJWjVEOwrJGk,23297 +pygments/lexers/ooc.py,sha256=kD1XaJZaihDF_s-Vyu1Bx68S_9zFt2rhox7NF8LpOZM,3002 +pygments/lexers/openscad.py,sha256=h9I1k8kiuQmhX5vZm6VDSr2fa5Finy0sN8ZDIE-jx1c,3700 +pygments/lexers/other.py,sha256=WLVyqPsvm9oSXIbZwbfyJloS6HGgoFW5nVTaU1uQpTw,1763 +pygments/lexers/parasail.py,sha256=DWMGhtyQgGTXbIgQl_mID6CKqi-Dhbvs_dTkmvrZXfE,2719 +pygments/lexers/parsers.py,sha256=feNgxroPoWRf0NEsON2mtmKDUfslIQppukw6ndEsQ3M,26596 +pygments/lexers/pascal.py,sha256=N2tRAjlXnTxggAzzk2tOOAVzeC2MBzrXy97_HQl5n44,30989 +pygments/lexers/pawn.py,sha256=LWUYQYsebMMt2d5oxX1HYWvBqbakR1h7Av_z8Vw94Wg,8253 +pygments/lexers/pddl.py,sha256=Mk4_BzlROJCd0xR4KKRRSrbj0F7LLQcBRjmsmtWmrCg,2989 +pygments/lexers/perl.py,sha256=9BXn3tyHMA49NvzbM9E2czSCHjeU7bvaPLUcoZrhz-4,39192 +pygments/lexers/phix.py,sha256=hZqychqo5sFMBDESzDPXg1DYHQe_9sn294UfbjihaFk,23249 +pygments/lexers/php.py,sha256=l4hzQrlm0525i5dSw9Vmjcai3TzbPT6DkjzxPg9l6Zc,13061 +pygments/lexers/pointless.py,sha256=WSDjqQyGrNIGmTCdaMxl4zk7OZTlJAMzeUZ02kfgcTI,1974 +pygments/lexers/pony.py,sha256=EXrMkacqMZblI7v4AvBRQe-3Py8__bx5FOgjCLdfXxQ,3279 +pygments/lexers/praat.py,sha256=4UFK-nbC6WkZBhJgcQqEGqq9CocJkW7AmT_OJQbjWzk,12676 +pygments/lexers/procfile.py,sha256=05W2fyofLTP-FbEdSXD1eles-PPqVNfF6RWXjQdW2us,1155 +pygments/lexers/prolog.py,sha256=9Kc5YNUFqkfWu2sYoyzC3RX65abf1bm7oHr86z1s4kQ,12866 +pygments/lexers/promql.py,sha256=n-0vo-o8-ZasqP3Va4ujs562UfZSLfZF-RzT71yL0Tk,4738 +pygments/lexers/prql.py,sha256=PFReuvhbv4K5aeu6lvDfw4m-3hULkB3r43bKAy948os,8747 +pygments/lexers/ptx.py,sha256=KSHAvbiNVUntKilQ6EPYoLFocmJpRsBy_7fW6_Nrs1Y,4501 +pygments/lexers/python.py,sha256=WZe7fBAHKZ_BxPg8qIU26UGhk8qwUYyENJ3IyPW64mc,53805 +pygments/lexers/q.py,sha256=WQFUh3JrpK2j-VGW_Ytn3uJ5frUNmQIFnLtMVGRA9DI,6936 +pygments/lexers/qlik.py,sha256=2wqwdfIjrAz6RNBsP4MyeLX8Z7QpIGzxtf1CvaOlr_g,3693 +pygments/lexers/qvt.py,sha256=XMBnsWRrvCDf989OuDeb-KpszAkeETiACyaghZeL1ns,6103 +pygments/lexers/r.py,sha256=B6WgrD9SY1UTCV1fQBSlZbezPfpYsARn3FQIHcFYOiM,6474 +pygments/lexers/rdf.py,sha256=qUzxLna9v071bHhZAjdsBi8dKaJNk_h9g1ZRUAYCfoo,16056 +pygments/lexers/rebol.py,sha256=4u3N4kzui55HapopXDu3Kt0jczxDZ4buzwR7Mt4tQiM,18259 +pygments/lexers/rego.py,sha256=Rx5Gphbktr9ojg5DbqlyxHeQqqtF7g8W-oF0rmloDNY,1748 +pygments/lexers/resource.py,sha256=ioEzgWksB5HCjoz85XNkQPSd7n5kL0SZiuPkJP1hunQ,2927 +pygments/lexers/ride.py,sha256=kCWdxuR3PclVi4wiA0uUx4CYEFwuTqoMsKjhSW4X3yg,5035 +pygments/lexers/rita.py,sha256=Mj1QNxx1sWAZYC02kw8piVckaiw9B0MqQtiIiDFH0pA,1127 +pygments/lexers/rnc.py,sha256=g7ZD334PMGUqy_Ij64laSN1vJerwHqVkegfMCa3E-y8,1972 +pygments/lexers/roboconf.py,sha256=HbYuK5CqmQdd63SRY2nle01r7-p7mil0SnoauYDmEOY,2074 +pygments/lexers/robotframework.py,sha256=c4U1B9Q9ITBCTohqJTZOvkfyeVbenN4xhzSWIoZh5eU,18448 +pygments/lexers/ruby.py,sha256=uG617E5abBZcECRCqkhIfc-IbZcRb5cGuUZq_xpax90,22753 +pygments/lexers/rust.py,sha256=ZY-9vtsreBP0NfDd0WCouLSp_9MChAL8U8Abe-m9PB8,8260 +pygments/lexers/sas.py,sha256=C1Uz2s9DU6_s2kL-cB_PAGPtpyK5THlmhNmCumC1l48,9456 +pygments/lexers/savi.py,sha256=jrmruK0GnXktgBTWXW3oN3TXtofn3HBbkMlHnR84cko,4878 +pygments/lexers/scdoc.py,sha256=DXRmFDmYuc7h3gPAAVhfcL1OEbNBK5RdPpJqQzF3ZTk,2524 +pygments/lexers/scripting.py,sha256=eaYlkDK-_cAwTcCBHP6QXBCz8n6OzbhzdkRe0uV0xWY,81814 +pygments/lexers/sgf.py,sha256=w6C513ENaO2YCnqrduK7k03NaMDf-pgygvfzq2NaSRk,1985 +pygments/lexers/shell.py,sha256=dCS1zwkf5KwTog4__MnMC7h3Xmwv4_d3fnEV29tSwXI,36381 +pygments/lexers/sieve.py,sha256=eob-L84yf2jmhdNyYZUlbUJozdcd6GXcHW68lmAe8WE,2514 +pygments/lexers/slash.py,sha256=I-cRepmaxhL1SgYvD1hHX3gNBFI8NPszdU7hn1o5JlA,8484 +pygments/lexers/smalltalk.py,sha256=ue2PmqDK2sw0j75WdseiiENJBdZ1OwysH2Op1QN1r24,7204 +pygments/lexers/smithy.py,sha256=VREWoeuz7ANap_Uiopn7rs0Tnsfc-xBisDJKRGQY_y8,2659 +pygments/lexers/smv.py,sha256=He_VBSMbWONMWZmkrB5RYR0cfHVnMyKIXz68IFYl-a8,2805 +pygments/lexers/snobol.py,sha256=qDzb41xQQWMNmjB2MtZs23pFoFgZ2gbRZhK_Ir03r7I,2778 +pygments/lexers/solidity.py,sha256=Tixfnwku4Yezj6nNm8xVaw7EdV1qgAgdwahdTFP0St8,3163 +pygments/lexers/soong.py,sha256=Vm18vV4g6T8UPgjjY2yTRlSXGDpZowmuqQUBFfm4A9A,2339 +pygments/lexers/sophia.py,sha256=2YtYIT8iwAoW0B7TZuuoG_ZILhJV-2A7oBGat-98naE,3376 +pygments/lexers/special.py,sha256=8JuR2Vex8X-RWnC36S0HXTHWp2qmZclc90-TrLUWyaY,3585 +pygments/lexers/spice.py,sha256=m4nK0q4Sq_OFQez7kGWfki0No4ZV24YrONfHVj1Piqs,2790 +pygments/lexers/sql.py,sha256=WSG6vOsR87EEEwSQefP_Z7TauUG_BjqMHUFmPaSOVj4,41476 +pygments/lexers/srcinfo.py,sha256=B8vDs-sJogG3mWa5Hp_7JfHHUMyYRwGvKv6cKbFQXLM,1746 +pygments/lexers/stata.py,sha256=Zr9BC52D5O_3BbdW0N-tzoUmy0NTguL2sC-saXRVM-c,6415 +pygments/lexers/supercollider.py,sha256=_H5wDrn0DiGnlhB_cz6Rt_lo2TvqjSm0o6NPTd9R4Ko,3697 +pygments/lexers/tablegen.py,sha256=1JjedXYY18BNiY9JtNGLOtGfiwduNDZpQLBGTeQ6jAw,3987 +pygments/lexers/tact.py,sha256=X_lsxjFUMaC1TmYysXJq9tmAGifRnil83Bt1zA86Xdo,10809 +pygments/lexers/tal.py,sha256=xS9PlaWQOPj8MVr56fUNq31vUQKRWoLTlyWj9ZHm8AM,2904 +pygments/lexers/tcl.py,sha256=lK97ju4nikkt-oGOzIeyFEM98yq4dZSI8uEmYsq0R6c,5512 +pygments/lexers/teal.py,sha256=t3dqy_Arwv8_yExbX_xiFxv1TqJLPv4vh1MVKjKwS4Y,3522 +pygments/lexers/templates.py,sha256=BVdjYeoacIUuFyHTG39j4PxeNCe5E1oUURjH1rITrI4,75731 +pygments/lexers/teraterm.py,sha256=ciwztagW5Drg2gr17Qykrh6GwMsKy7e4xdQshX95GyQ,9718 +pygments/lexers/testing.py,sha256=YZgDgUEaLEYKSKEqpDsUi3Bn-Db_D42IlyiSsr1oX8U,10810 +pygments/lexers/text.py,sha256=nOCQPssIlKdVWU3PKxZiBPkf_KFM2V48IOssSyqhFY8,1068 +pygments/lexers/textedit.py,sha256=ttT4Ph-hIdgFLG6maRy_GskkziTFK0Wcg28yU0s6lek,7760 +pygments/lexers/textfmts.py,sha256=mi9KLEq4mrzDJbEc8G3VM-mSki_Tylkzodu47yH6z84,15524 +pygments/lexers/theorem.py,sha256=51ppBAEdhJmwU_lC916zMyjEoKLXqf89VAE_Lr0PNCc,17855 +pygments/lexers/thingsdb.py,sha256=x_fHNkLA-hIJyeIs6rg_X8n5OLYvFqaSu1FhI3apI5Y,6017 +pygments/lexers/tlb.py,sha256=ue2gqm45BI512lM13O8skAky9zAb7pLMrxZ8pbt5zRU,1450 +pygments/lexers/tls.py,sha256=_uQUVuMRDOhN-XUyGR5DIlVCk1CUZ1fIOSN4_WQYPKk,1540 +pygments/lexers/tnt.py,sha256=pK4LgoKON7u1xF66JYFncAPSbD8DZaeI_WTZ9HqEFlY,10456 +pygments/lexers/trafficscript.py,sha256=X3B8kgxS54ecuok9ic6Hkp-UMn5DvOmCK0p70Tz27Cw,1506 +pygments/lexers/typoscript.py,sha256=mBuePiVZUoAORPKsHwrx6fBWiy3fAIqG-2O67QmMiFI,8332 +pygments/lexers/typst.py,sha256=zIJBEhUXtWp5OiyAmvFA5m8d1EQG-ocwrJ677dvTUAk,7167 +pygments/lexers/ul4.py,sha256=rCaw0J9j3cdql9lX_HTilg65k9-9S118zOA6TAYfxaM,10499 +pygments/lexers/unicon.py,sha256=RAqoCnAAJBYOAGdR8ng0g6FtB39bGemLRlIqv5mcg9E,18625 +pygments/lexers/urbi.py,sha256=ajNP70NJg32jNnFDZsLvr_-4TToSGqRGkFyAPIJLfCU,6082 +pygments/lexers/usd.py,sha256=2eEGouolodYS402P_gtBrn4lLzpg1z8uHwPCKqjUb_k,3304 +pygments/lexers/varnish.py,sha256=dSh0Ku9SrjmlB29Fi_mWdWavN7M0cMKeepR4a34sOyI,7473 +pygments/lexers/verification.py,sha256=Qu433Q_h3EK3uS4bJoLRFZK0kIVwzX5AFKsa4Z-qnxA,3934 +pygments/lexers/verifpal.py,sha256=buyOOzCo_dGnoC40h0tthylHVVpgDt8qXu4olLvYy_4,2661 +pygments/lexers/vip.py,sha256=2lEV4cLV9p4E37wctBL7zkZ4ZU4p3HVsiLJFzB1bie0,5711 +pygments/lexers/vyper.py,sha256=Zq6sQIUBk6mBdpgOVgu3A6swGoBne0kDlRyjZznm2BY,5615 +pygments/lexers/web.py,sha256=4W9a7vcskrGJnxt4KmoE3SZydWB1qLq7lP2XS85J_m8,913 +pygments/lexers/webassembly.py,sha256=zgcMouzLawcbeFr6w_SOvGoUR68ZtqnnsbOcWEVleLk,5698 +pygments/lexers/webidl.py,sha256=ODtVmw4gVzI8HQWxuEckP6KMwm8WP2G2lSZEjagDXts,10516 +pygments/lexers/webmisc.py,sha256=-_-INDVdk47e2jlj-9bFcuLtntqVorBqIjlnwPfZFdI,40564 +pygments/lexers/wgsl.py,sha256=9igd9dzixGIgNewruv9mPnFms-c9BahkZcCCrZygv84,11880 +pygments/lexers/whiley.py,sha256=lMr750lA4MZsB4xqzVsIRtVMJIC3_dArhFYTHvOPwvA,4017 +pygments/lexers/wowtoc.py,sha256=8xxvf0xGeYtf4PE7KtkHZ_ly9xY_XXHrpCitdKE42Ro,4076 +pygments/lexers/wren.py,sha256=goGXnAMKKa13LLL40ybT3aMGPrk3gCRwZQFYAkKB_w0,3229 +pygments/lexers/x10.py,sha256=Q-AmgdF2E-N7mtOPpZ07CsxrTVnikyqC4uRRv6H75sk,1943 +pygments/lexers/xorg.py,sha256=9ttrBd3_Y2nXANsqtMposSgblYmMYqWXQ-Iz5RH9RsU,925 +pygments/lexers/yang.py,sha256=13CWbSaNr9giOHz4o0SXSklh0bfWt0ah14jJGpTvcn0,4499 +pygments/lexers/yara.py,sha256=jUSv78KTDfguCoAoAZKbYzQERkkyxBBWv5dInVrkDxo,2427 +pygments/lexers/zig.py,sha256=f-80MVOSp1KnczAMokQLVM-_wAEOD16EcGFnaCNlsN0,3976 +pygments/modeline.py,sha256=K5eSkR8GS1r5OkXXTHOcV0aM_6xpk9eWNEIAW-OOJ2g,1005 +pygments/plugin.py,sha256=tPx0rJCTIZ9ioRgLNYG4pifCbAwTRUZddvLw-NfAk2w,1891 +pygments/regexopt.py,sha256=wXaP9Gjp_hKAdnICqoDkRxAOQJSc4v3X6mcxx3z-TNs,3072 +pygments/scanner.py,sha256=nNcETRR1tRuiTaHmHSTTECVYFPcLf6mDZu1e4u91A9E,3092 +pygments/sphinxext.py,sha256=VEe_oHNgLoEGMHc2ROfbee2mF2PPREFyE6_m_JN5FvQ,7898 +pygments/style.py,sha256=Cpw9dCAyW3_JAwFRXOJXmtKb5ZwO2_5KSmlq6q4fZw4,6408 +pygments/styles/__init__.py,sha256=f9KCQXN4uKbe8aI8-L3qTC-_XPfT563FwTg6VTGVfwI,2006 +pygments/styles/__pycache__/__init__.cpython-312.pyc,, +pygments/styles/__pycache__/_mapping.cpython-312.pyc,, +pygments/styles/__pycache__/abap.cpython-312.pyc,, +pygments/styles/__pycache__/algol.cpython-312.pyc,, +pygments/styles/__pycache__/algol_nu.cpython-312.pyc,, +pygments/styles/__pycache__/arduino.cpython-312.pyc,, +pygments/styles/__pycache__/autumn.cpython-312.pyc,, +pygments/styles/__pycache__/borland.cpython-312.pyc,, +pygments/styles/__pycache__/bw.cpython-312.pyc,, +pygments/styles/__pycache__/coffee.cpython-312.pyc,, +pygments/styles/__pycache__/colorful.cpython-312.pyc,, +pygments/styles/__pycache__/default.cpython-312.pyc,, +pygments/styles/__pycache__/dracula.cpython-312.pyc,, +pygments/styles/__pycache__/emacs.cpython-312.pyc,, +pygments/styles/__pycache__/friendly.cpython-312.pyc,, +pygments/styles/__pycache__/friendly_grayscale.cpython-312.pyc,, +pygments/styles/__pycache__/fruity.cpython-312.pyc,, +pygments/styles/__pycache__/gh_dark.cpython-312.pyc,, +pygments/styles/__pycache__/gruvbox.cpython-312.pyc,, +pygments/styles/__pycache__/igor.cpython-312.pyc,, +pygments/styles/__pycache__/inkpot.cpython-312.pyc,, +pygments/styles/__pycache__/lightbulb.cpython-312.pyc,, +pygments/styles/__pycache__/lilypond.cpython-312.pyc,, +pygments/styles/__pycache__/lovelace.cpython-312.pyc,, +pygments/styles/__pycache__/manni.cpython-312.pyc,, +pygments/styles/__pycache__/material.cpython-312.pyc,, +pygments/styles/__pycache__/monokai.cpython-312.pyc,, +pygments/styles/__pycache__/murphy.cpython-312.pyc,, +pygments/styles/__pycache__/native.cpython-312.pyc,, +pygments/styles/__pycache__/nord.cpython-312.pyc,, +pygments/styles/__pycache__/onedark.cpython-312.pyc,, +pygments/styles/__pycache__/paraiso_dark.cpython-312.pyc,, +pygments/styles/__pycache__/paraiso_light.cpython-312.pyc,, +pygments/styles/__pycache__/pastie.cpython-312.pyc,, +pygments/styles/__pycache__/perldoc.cpython-312.pyc,, +pygments/styles/__pycache__/rainbow_dash.cpython-312.pyc,, +pygments/styles/__pycache__/rrt.cpython-312.pyc,, +pygments/styles/__pycache__/sas.cpython-312.pyc,, +pygments/styles/__pycache__/solarized.cpython-312.pyc,, +pygments/styles/__pycache__/staroffice.cpython-312.pyc,, +pygments/styles/__pycache__/stata_dark.cpython-312.pyc,, +pygments/styles/__pycache__/stata_light.cpython-312.pyc,, +pygments/styles/__pycache__/tango.cpython-312.pyc,, +pygments/styles/__pycache__/trac.cpython-312.pyc,, +pygments/styles/__pycache__/vim.cpython-312.pyc,, +pygments/styles/__pycache__/vs.cpython-312.pyc,, +pygments/styles/__pycache__/xcode.cpython-312.pyc,, +pygments/styles/__pycache__/zenburn.cpython-312.pyc,, +pygments/styles/_mapping.py,sha256=6lovFUE29tz6EsV3XYY4hgozJ7q1JL7cfO3UOlgnS8w,3312 +pygments/styles/abap.py,sha256=64Uwr8uPdEdcT-tE-Y2VveTXfH3SkqH9qdMgY49YHQI,749 +pygments/styles/algol.py,sha256=fCuk8ITTehvbJSufiaKlgnFsKbl-xFxxR82xhltc-cQ,2262 +pygments/styles/algol_nu.py,sha256=Gv9WfHJvYegGcUk1zcufQgsdXPNjCUNk8sAHyrSGGh4,2283 +pygments/styles/arduino.py,sha256=NoUB8xk7M1HGPoLfuySOLU0sVwoTuLcZqllXl2EO_iE,4557 +pygments/styles/autumn.py,sha256=fLLfjHXjxCl6crBAxEsBLH372ALMkFacA2bG6KFbJi4,2195 +pygments/styles/borland.py,sha256=_0ySKp4KGCSgtYjPe8uzD6gQhlmAIR4T43i-FoRYNOM,1611 +pygments/styles/bw.py,sha256=vhk8Xoj64fLPdA9IQU6mUVsYMel255jR-FDU7BjIHtI,1406 +pygments/styles/coffee.py,sha256=NqLt-fc7LONma1BGggbceVRY9uDE70WBuZXqK4zwaco,2308 +pygments/styles/colorful.py,sha256=mYcSbehtH7itH_QV9NqJp4Wna1X4lrwl2wkVXS2u-5A,2832 +pygments/styles/default.py,sha256=RTgG2zKWWUxPTDCFxhTnyZI_WZBIVgu5XsUpNvFisCA,2588 +pygments/styles/dracula.py,sha256=vRJmixBoSKV9o8NVQhXGViQqchhIYugfikLmvX0DoBw,2182 +pygments/styles/emacs.py,sha256=TiOG9oc83qToMCRMnJrXtWYqnzAqYycRz_50OoCKtxc,2535 +pygments/styles/friendly.py,sha256=oAi-l9anQTs9STDmUzXGDlOegatEOH4hpD0j6o6dZGM,2604 +pygments/styles/friendly_grayscale.py,sha256=a7Cqkzt6-uTiXvj6GoYBXzRvX5_zviCjjRB04Kf_-Q0,2828 +pygments/styles/fruity.py,sha256=GfSUTG0stlJr5Ow_saCaxbI2IB4-34Dp2TuRTpfUJBs,1324 +pygments/styles/gh_dark.py,sha256=ruNX3d4rf22rx-8HnwvGbNbXRQpXCNcHU1HNq6N4uNg,3590 +pygments/styles/gruvbox.py,sha256=KrFoHEoVnZW6XM9udyXncPomeGyZgIDsNWOH3kCrxFQ,3387 +pygments/styles/igor.py,sha256=fYYPhM0dRCvcDTMVrMVO5oFKnYm-8YVlsuVBoczFLtY,737 +pygments/styles/inkpot.py,sha256=jggSeX9NV15eOL2oJaVmZ6vmV7LWRzXJQRUqcWEqGRs,2404 +pygments/styles/lightbulb.py,sha256=Y8u1qdvlHfBqI2jJex55SkvVatVo_FjEUzE6h-X7m-0,3172 +pygments/styles/lilypond.py,sha256=Y6fp_sEL-zESmxAaMxzjtrKk90cuDC_DalNdC8wj0nw,2066 +pygments/styles/lovelace.py,sha256=cA9uhmbnzY04MccsiYSgMY7fvb4WMRbegWBUrGvXh1M,3178 +pygments/styles/manni.py,sha256=g9FyO7plTwfMm2cU4iiKgdlkMlvQLG6l2Lwkgz5ITS4,2443 +pygments/styles/material.py,sha256=LDmgomAbgtJDZhbv446_zIwgYh50UAqEEtgYNUns1rQ,4201 +pygments/styles/monokai.py,sha256=lrxTJpkBarV9gTLkBQryZ6oNSjekAVheJueKJP5iEYA,5184 +pygments/styles/murphy.py,sha256=-AKZiLkpiWej-otjHMsYCE-I-_IzCOLJY-_GBdKRZRw,2805 +pygments/styles/native.py,sha256=l6tezGSQTB8p_SyOXJ0PWI7KzCeEdtsPmVc4Yn4_CwU,2043 +pygments/styles/nord.py,sha256=GDt3WAaqaWsiCeqpIBPxd8TEUX708fGfwaA7S0w0oy0,5391 +pygments/styles/onedark.py,sha256=k80cZEppCEF-HLoxy_FEA0QmQDZze68nHVMNGyUVa28,1719 +pygments/styles/paraiso_dark.py,sha256=Jkrg4nUKIVNF8U4fPNV_Smq_g9NFbb9eiUrjYpVgQZg,5662 +pygments/styles/paraiso_light.py,sha256=MxN964ZEpze3wF0ss-igaa2I7E684MHe-Zq0rWPH3wo,5668 +pygments/styles/pastie.py,sha256=ZvAs9UpBNYFC-5PFrCRGYnm3FoPKb-eKR-ozbWZP-4g,2525 +pygments/styles/perldoc.py,sha256=HSxB93e4UpQkZspReQ34FeJbZ-59ksGvdaH-hToehi8,2230 +pygments/styles/rainbow_dash.py,sha256=4ugL18Or7aNtaLfPfCLFRiFy0Gu2RA4a9G2LQUE9SrM,2390 +pygments/styles/rrt.py,sha256=fgzfpC0PC_SCcLOMCNEIQTjPUMOncRe7SR10GfSRbXY,1006 +pygments/styles/sas.py,sha256=yzoXmbfQ2ND1WWq93b4vVGYkQSZHPqb4ymes9YYRT3w,1440 +pygments/styles/solarized.py,sha256=qupILFZn02WspnAF5SPYb-W8guo9xnUtjb1HeLw3XgE,4247 +pygments/styles/staroffice.py,sha256=CLbBeMoxay21Xyu3Af2p4xUXyG1_6ydCbvs5RJKYe5w,831 +pygments/styles/stata_dark.py,sha256=vX8SwHV__sG92F4CKribG08MJfSVq98dgs7gEA_n9yc,1257 +pygments/styles/stata_light.py,sha256=uV3GE-ylvffQ0yN3py1YAVqBB5wflIKZbceyK1Lqvrc,1289 +pygments/styles/tango.py,sha256=O2wcM4hHuU1Yt071M9CK7JPtiiSCqyxtT9tbiQICV28,7137 +pygments/styles/trac.py,sha256=9kMv1ZZyMKACWlx2fQVjRP0I2pgcRYCNrd7iGGZg9qk,1981 +pygments/styles/vim.py,sha256=J7_TqvrGkTX_XuTHW0In5wqPLAUPRWyr1122XueZWmM,2019 +pygments/styles/vs.py,sha256=s7YnzbIPuFU3LIke27mc4lAQSn2R3vbbHc1baMGSU_U,1130 +pygments/styles/xcode.py,sha256=PbQdzgGaA4a9LAU1i58alY9kM4IFlQX5jHQwOYmf_Rk,1504 +pygments/styles/zenburn.py,sha256=suZEKzBTCYdhf2cxNwcY7UATJK1tq5eYhGdBcXdf6MU,2203 +pygments/token.py,sha256=WbdWGhYm_Vosb0DDxW9lHNPgITXfWTsQmHt6cy9RbcM,6226 +pygments/unistring.py,sha256=al-_rBemRuGvinsrM6atNsHTmJ6DUbw24q2O2Ru1cBc,63208 +pygments/util.py,sha256=oRtSpiAo5jM9ulntkvVbgXUdiAW57jnuYGB7t9fYuhc,10031 diff --git a/lib/python3.12/site-packages/pygments-2.19.2.dist-info/WHEEL b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/WHEEL new file mode 100644 index 0000000000000000000000000000000000000000..12228d414b6cfed7c39d3781c85c63256a1d7fb5 --- /dev/null +++ b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/WHEEL @@ -0,0 +1,4 @@ +Wheel-Version: 1.0 +Generator: hatchling 1.27.0 +Root-Is-Purelib: true +Tag: py3-none-any diff --git a/lib/python3.12/site-packages/pygments-2.19.2.dist-info/entry_points.txt b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/entry_points.txt new file mode 100644 index 0000000000000000000000000000000000000000..15498e35f53320bd5e1de176928daabf26be0109 --- /dev/null +++ b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/entry_points.txt @@ -0,0 +1,2 @@ +[console_scripts] +pygmentize = pygments.cmdline:main diff --git a/lib/python3.12/site-packages/pygments-2.19.2.dist-info/licenses/AUTHORS b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/licenses/AUTHORS new file mode 100644 index 0000000000000000000000000000000000000000..811c66ae178b354c0a0eb9bd5e7a9ac63b4b056c --- /dev/null +++ b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/licenses/AUTHORS @@ -0,0 +1,291 @@ +Pygments is written and maintained by Georg Brandl . + +Major developers are Tim Hatch and Armin Ronacher +. + +Other contributors, listed alphabetically, are: + +* Sam Aaron -- Ioke lexer +* Jean Abou Samra -- LilyPond lexer +* João Abecasis -- JSLT lexer +* Ali Afshar -- image formatter +* Thomas Aglassinger -- Easytrieve, JCL, Rexx, Transact-SQL and VBScript + lexers +* Maxence Ahlouche -- PostgreSQL Explain lexer +* Muthiah Annamalai -- Ezhil lexer +* Nikolay Antipov -- OpenSCAD lexer +* Kumar Appaiah -- Debian control lexer +* Andreas Amann -- AppleScript lexer +* Timothy Armstrong -- Dart lexer fixes +* Jeffrey Arnold -- R/S, Rd, BUGS, Jags, and Stan lexers +* Eiríkr Åsheim -- Uxntal lexer +* Jeremy Ashkenas -- CoffeeScript lexer +* José Joaquín Atria -- Praat lexer +* Stefan Matthias Aust -- Smalltalk lexer +* Lucas Bajolet -- Nit lexer +* Ben Bangert -- Mako lexers +* Max Battcher -- Darcs patch lexer +* Thomas Baruchel -- APL lexer +* Tim Baumann -- (Literate) Agda lexer +* Paul Baumgart, 280 North, Inc. -- Objective-J lexer +* Michael Bayer -- Myghty lexers +* Thomas Beale -- Archetype lexers +* John Benediktsson -- Factor lexer +* David Benjamin, Google LLC -- TLS lexer +* Trevor Bergeron -- mIRC formatter +* Vincent Bernat -- LessCSS lexer +* Christopher Bertels -- Fancy lexer +* Sébastien Bigaret -- QVT Operational lexer +* Jarrett Billingsley -- MiniD lexer +* Adam Blinkinsop -- Haskell, Redcode lexers +* Stéphane Blondon -- Procfile, SGF and Sieve lexers +* Frits van Bommel -- assembler lexers +* Pierre Bourdon -- bugfixes +* Martijn Braam -- Kernel log lexer, BARE lexer +* JD Browne, Google LLC -- GoogleSQL lexer +* Matthias Bussonnier -- ANSI style handling for terminal-256 formatter +* chebee7i -- Python traceback lexer improvements +* Hiram Chirino -- Scaml and Jade lexers +* Mauricio Caceres -- SAS and Stata lexers. +* Michael Camilleri, John Gabriele, sogaiu -- Janet lexer +* Daren Chandisingh -- Gleam lexer +* Ian Cooper -- VGL lexer +* David Corbett -- Inform, Jasmin, JSGF, Snowball, and TADS 3 lexers +* Leaf Corcoran -- MoonScript lexer +* Fraser Cormack -- TableGen lexer +* Gabriel Corona -- ASN.1 lexer +* Christopher Creutzig -- MuPAD lexer +* Daniël W. Crompton -- Pike lexer +* Pete Curry -- bugfixes +* Bryan Davis -- EBNF lexer +* Bruno Deferrari -- Shen lexer +* Walter Dörwald -- UL4 lexer +* Luke Drummond -- Meson lexer +* Giedrius Dubinskas -- HTML formatter improvements +* Owen Durni -- Haxe lexer +* Alexander Dutton, Oxford University Computing Services -- SPARQL lexer +* James Edwards -- Terraform lexer +* Nick Efford -- Python 3 lexer +* Sven Efftinge -- Xtend lexer +* Artem Egorkine -- terminal256 formatter +* Matthew Fernandez -- CAmkES lexer +* Paweł Fertyk -- GDScript lexer, HTML formatter improvements +* Michael Ficarra -- CPSA lexer +* James H. Fisher -- PostScript lexer +* Amanda Fitch, Google LLC -- GoogleSQL lexer +* William S. Fulton -- SWIG lexer +* Carlos Galdino -- Elixir and Elixir Console lexers +* Michael Galloy -- IDL lexer +* Naveen Garg -- Autohotkey lexer +* Simon Garnotel -- FreeFem++ lexer +* Laurent Gautier -- R/S lexer +* Alex Gaynor -- PyPy log lexer +* Richard Gerkin -- Igor Pro lexer +* Alain Gilbert -- TypeScript lexer +* Alex Gilding -- BlitzBasic lexer +* GitHub, Inc -- DASM16, Augeas, TOML, and Slash lexers +* Bertrand Goetzmann -- Groovy lexer +* Krzysiek Goj -- Scala lexer +* Rostyslav Golda -- FloScript lexer +* Andrey Golovizin -- BibTeX lexers +* Matt Good -- Genshi, Cheetah lexers +* Michał Górny -- vim modeline support +* Alex Gosse -- TrafficScript lexer +* Patrick Gotthardt -- PHP namespaces support +* Hubert Gruniaux -- C and C++ lexer improvements +* Olivier Guibe -- Asymptote lexer +* Phil Hagelberg -- Fennel lexer +* Florian Hahn -- Boogie lexer +* Martin Harriman -- SNOBOL lexer +* Matthew Harrison -- SVG formatter +* Steven Hazel -- Tcl lexer +* Dan Michael Heggø -- Turtle lexer +* Aslak Hellesøy -- Gherkin lexer +* Greg Hendershott -- Racket lexer +* Justin Hendrick -- ParaSail lexer +* Jordi Gutiérrez Hermoso -- Octave lexer +* David Hess, Fish Software, Inc. -- Objective-J lexer +* Ken Hilton -- Typographic Number Theory and Arrow lexers +* Varun Hiremath -- Debian control lexer +* Rob Hoelz -- Perl 6 lexer +* Doug Hogan -- Mscgen lexer +* Ben Hollis -- Mason lexer +* Max Horn -- GAP lexer +* Fred Hornsey -- OMG IDL Lexer +* Alastair Houghton -- Lexer inheritance facility +* Tim Howard -- BlitzMax lexer +* Dustin Howett -- Logos lexer +* Ivan Inozemtsev -- Fantom lexer +* Hiroaki Itoh -- Shell console rewrite, Lexers for PowerShell session, + MSDOS session, BC, WDiff +* Brian R. Jackson -- Tea lexer +* Christian Jann -- ShellSession lexer +* Jonas Camillus Jeppesen -- Line numbers and line highlighting for + RTF-formatter +* Dennis Kaarsemaker -- sources.list lexer +* Dmitri Kabak -- Inferno Limbo lexer +* Igor Kalnitsky -- vhdl lexer +* Colin Kennedy - USD lexer +* Alexander Kit -- MaskJS lexer +* Pekka Klärck -- Robot Framework lexer +* Gerwin Klein -- Isabelle lexer +* Eric Knibbe -- Lasso lexer +* Stepan Koltsov -- Clay lexer +* Oliver Kopp - Friendly grayscale style +* Adam Koprowski -- Opa lexer +* Benjamin Kowarsch -- Modula-2 lexer +* Domen Kožar -- Nix lexer +* Oleh Krekel -- Emacs Lisp lexer +* Alexander Kriegisch -- Kconfig and AspectJ lexers +* Marek Kubica -- Scheme lexer +* Jochen Kupperschmidt -- Markdown processor +* Gerd Kurzbach -- Modelica lexer +* Jon Larimer, Google Inc. -- Smali lexer +* Olov Lassus -- Dart lexer +* Matt Layman -- TAP lexer +* Dan Lazin, Google LLC -- GoogleSQL lexer +* Kristian Lyngstøl -- Varnish lexers +* Sylvestre Ledru -- Scilab lexer +* Chee Sing Lee -- Flatline lexer +* Mark Lee -- Vala lexer +* Thomas Linder Puls -- Visual Prolog lexer +* Pete Lomax -- Phix lexer +* Valentin Lorentz -- C++ lexer improvements +* Ben Mabey -- Gherkin lexer +* Angus MacArthur -- QML lexer +* Louis Mandel -- X10 lexer +* Louis Marchand -- Eiffel lexer +* Simone Margaritelli -- Hybris lexer +* Tim Martin - World of Warcraft TOC lexer +* Kirk McDonald -- D lexer +* Gordon McGregor -- SystemVerilog lexer +* Stephen McKamey -- Duel/JBST lexer +* Brian McKenna -- F# lexer +* Charles McLaughlin -- Puppet lexer +* Kurt McKee -- Tera Term macro lexer, PostgreSQL updates, MySQL overhaul, JSON lexer +* Joe Eli McIlvain -- Savi lexer +* Lukas Meuser -- BBCode formatter, Lua lexer +* Cat Miller -- Pig lexer +* Paul Miller -- LiveScript lexer +* Hong Minhee -- HTTP lexer +* Michael Mior -- Awk lexer +* Bruce Mitchener -- Dylan lexer rewrite +* Reuben Morais -- SourcePawn lexer +* Jon Morton -- Rust lexer +* Paulo Moura -- Logtalk lexer +* Mher Movsisyan -- DTD lexer +* Dejan Muhamedagic -- Crmsh lexer +* Adrien Nayrat -- PostgreSQL Explain lexer +* Ana Nelson -- Ragel, ANTLR, R console lexers +* David Neto, Google LLC -- WebGPU Shading Language lexer +* Kurt Neufeld -- Markdown lexer +* Nam T. Nguyen -- Monokai style +* Jesper Noehr -- HTML formatter "anchorlinenos" +* Mike Nolta -- Julia lexer +* Avery Nortonsmith -- Pointless lexer +* Jonas Obrist -- BBCode lexer +* Edward O'Callaghan -- Cryptol lexer +* David Oliva -- Rebol lexer +* Pat Pannuto -- nesC lexer +* Jon Parise -- Protocol buffers and Thrift lexers +* Benjamin Peterson -- Test suite refactoring +* Ronny Pfannschmidt -- BBCode lexer +* Dominik Picheta -- Nimrod lexer +* Andrew Pinkham -- RTF Formatter Refactoring +* Clément Prévost -- UrbiScript lexer +* Tanner Prynn -- cmdline -x option and loading lexers from files +* Oleh Prypin -- Crystal lexer (based on Ruby lexer) +* Nick Psaris -- K and Q lexers +* Xidorn Quan -- Web IDL lexer +* Elias Rabel -- Fortran fixed form lexer +* raichoo -- Idris lexer +* Daniel Ramirez -- GDScript lexer +* Kashif Rasul -- CUDA lexer +* Nathan Reed -- HLSL lexer +* Justin Reidy -- MXML lexer +* Jonathon Reinhart, Google LLC -- Soong lexer +* Norman Richards -- JSON lexer +* Corey Richardson -- Rust lexer updates +* Fabrizio Riguzzi -- cplint leder +* Lubomir Rintel -- GoodData MAQL and CL lexers +* Andre Roberge -- Tango style +* Georg Rollinger -- HSAIL lexer +* Michiel Roos -- TypoScript lexer +* Konrad Rudolph -- LaTeX formatter enhancements +* Mario Ruggier -- Evoque lexers +* Miikka Salminen -- Lovelace style, Hexdump lexer, lexer enhancements +* Stou Sandalski -- NumPy, FORTRAN, tcsh and XSLT lexers +* Matteo Sasso -- Common Lisp lexer +* Joe Schafer -- Ada lexer +* Max Schillinger -- TiddlyWiki5 lexer +* Andrew Schmidt -- X++ lexer +* Ken Schutte -- Matlab lexers +* René Schwaiger -- Rainbow Dash style +* Sebastian Schweizer -- Whiley lexer +* Tassilo Schweyer -- Io, MOOCode lexers +* Pablo Seminario -- PromQL lexer +* Ted Shaw -- AutoIt lexer +* Joerg Sieker -- ABAP lexer +* Robert Simmons -- Standard ML lexer +* Kirill Simonov -- YAML lexer +* Corbin Simpson -- Monte lexer +* Ville Skyttä -- ASCII armored lexer +* Alexander Smishlajev -- Visual FoxPro lexer +* Steve Spigarelli -- XQuery lexer +* Jerome St-Louis -- eC lexer +* Camil Staps -- Clean and NuSMV lexers; Solarized style +* James Strachan -- Kotlin lexer +* Tom Stuart -- Treetop lexer +* Colin Sullivan -- SuperCollider lexer +* Ben Swift -- Extempore lexer +* tatt61880 -- Kuin lexer +* Edoardo Tenani -- Arduino lexer +* Tiberius Teng -- default style overhaul +* Jeremy Thurgood -- Erlang, Squid config lexers +* Brian Tiffin -- OpenCOBOL lexer +* Bob Tolbert -- Hy lexer +* Doug Torrance -- Macaulay2 lexer +* Matthias Trute -- Forth lexer +* Tuoa Spi T4 -- Bdd lexer +* Erick Tryzelaar -- Felix lexer +* Alexander Udalov -- Kotlin lexer improvements +* Thomas Van Doren -- Chapel lexer +* Dave Van Ee -- Uxntal lexer updates +* Daniele Varrazzo -- PostgreSQL lexers +* Abe Voelker -- OpenEdge ABL lexer +* Pepijn de Vos -- HTML formatter CTags support +* Matthias Vallentin -- Bro lexer +* Benoît Vinot -- AMPL lexer +* Linh Vu Hong -- RSL lexer +* Taavi Väänänen -- Debian control lexer +* Immanuel Washington -- Smithy lexer +* Nathan Weizenbaum -- Haml and Sass lexers +* Nathan Whetsell -- Csound lexers +* Dietmar Winkler -- Modelica lexer +* Nils Winter -- Smalltalk lexer +* Davy Wybiral -- Clojure lexer +* Whitney Young -- ObjectiveC lexer +* Diego Zamboni -- CFengine3 lexer +* Enrique Zamudio -- Ceylon lexer +* Alex Zimin -- Nemerle lexer +* Rob Zimmerman -- Kal lexer +* Evgenii Zheltonozhskii -- Maple lexer +* Vincent Zurczak -- Roboconf lexer +* Hubert Gruniaux -- C and C++ lexer improvements +* Thomas Symalla -- AMDGPU Lexer +* 15b3 -- Image Formatter improvements +* Fabian Neumann -- CDDL lexer +* Thomas Duboucher -- CDDL lexer +* Philipp Imhof -- Pango Markup formatter +* Thomas Voss -- Sed lexer +* Martin Fischer -- WCAG contrast testing +* Marc Auberer -- Spice lexer +* Amr Hesham -- Carbon lexer +* diskdance -- Wikitext lexer +* vanillajonathan -- PRQL lexer +* Nikolay Antipov -- OpenSCAD lexer +* Markus Meyer, Nextron Systems -- YARA lexer +* Hannes Römer -- Mojo lexer +* Jan Frederik Schaefer -- PDDL lexer + +Many thanks for all contributions! diff --git a/lib/python3.12/site-packages/pygments-2.19.2.dist-info/licenses/LICENSE b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/licenses/LICENSE new file mode 100644 index 0000000000000000000000000000000000000000..446a1a805c8a949579fb8a9799f2bec7777349dc --- /dev/null +++ b/lib/python3.12/site-packages/pygments-2.19.2.dist-info/licenses/LICENSE @@ -0,0 +1,25 @@ +Copyright (c) 2006-2022 by the respective authors (see AUTHORS file). +All rights reserved. + +Redistribution and use in source and binary forms, with or without +modification, are permitted provided that the following conditions are +met: + +* Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + +* Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS +"AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT +LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR +A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT +OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, +SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT +LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, +DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY +THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT +(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE +OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. diff --git a/lib/python3.12/site-packages/termcolor/__init__.py b/lib/python3.12/site-packages/termcolor/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..8ef49cf884e45f6c610207e66704444a760b2f55 --- /dev/null +++ b/lib/python3.12/site-packages/termcolor/__init__.py @@ -0,0 +1,14 @@ +"""ANSI color formatting for output in terminal.""" + +from __future__ import annotations + +from termcolor.termcolor import ATTRIBUTES, COLORS, HIGHLIGHTS, RESET, colored, cprint + +__all__ = [ + "ATTRIBUTES", + "COLORS", + "HIGHLIGHTS", + "RESET", + "colored", + "cprint", +] diff --git a/lib/python3.12/site-packages/termcolor/__main__.py b/lib/python3.12/site-packages/termcolor/__main__.py new file mode 100644 index 0000000000000000000000000000000000000000..21ae4c6c90909f24697ff6d52a3a11f08e3f7d92 --- /dev/null +++ b/lib/python3.12/site-packages/termcolor/__main__.py @@ -0,0 +1,86 @@ +from __future__ import annotations + +import os + +from termcolor import cprint + +if __name__ == "__main__": + print(f"Current terminal type: {os.getenv('TERM')}") + print("Test basic colors:") + cprint("Black color", "black") + cprint("Red color", "red") + cprint("Green color", "green") + cprint("Yellow color", "yellow") + cprint("Blue color", "blue") + cprint("Magenta color", "magenta") + cprint("Cyan color", "cyan") + cprint("White color", "white") + cprint("Light grey color", "light_grey") + cprint("Dark grey color", "dark_grey") + cprint("Light red color", "light_red") + cprint("Light green color", "light_green") + cprint("Light yellow color", "light_yellow") + cprint("Light blue color", "light_blue") + cprint("Light magenta color", "light_magenta") + cprint("Light cyan color", "light_cyan") + print("-" * 78) + + print("Test highlights:") + cprint("On black color", on_color="on_black") + cprint("On red color", on_color="on_red") + cprint("On green color", on_color="on_green") + cprint("On yellow color", on_color="on_yellow") + cprint("On blue color", on_color="on_blue") + cprint("On magenta color", on_color="on_magenta") + cprint("On cyan color", on_color="on_cyan") + cprint("On white color", color="black", on_color="on_white") + cprint("On light grey color", on_color="on_light_grey") + cprint("On dark grey color", on_color="on_dark_grey") + cprint("On light red color", on_color="on_light_red") + cprint("On light green color", on_color="on_light_green") + cprint("On light yellow color", on_color="on_light_yellow") + cprint("On light blue color", on_color="on_light_blue") + cprint("On light magenta color", on_color="on_light_magenta") + cprint("On light cyan color", on_color="on_light_cyan") + print("-" * 78) + + print("Test attributes:") + cprint("Bold black color", "black", attrs=["bold"]) + cprint("Dark red color", "red", attrs=["dark"]) + cprint("Underline green color", "green", attrs=["underline"]) + cprint("Blink yellow color", "yellow", attrs=["blink"]) + cprint("Reversed blue color", "blue", attrs=["reverse"]) + cprint("Concealed magenta color", "magenta", attrs=["concealed"]) + cprint("Strike red color", "red", attrs=["strike"]) + cprint( + "Bold underline reverse cyan color", + "cyan", + attrs=["bold", "underline", "reverse"], + ) + cprint( + "Dark blink concealed white color", + "white", + attrs=["dark", "blink", "concealed"], + ) + print("-" * 78) + + print("Test mixing:") + cprint("Underline red on black color", "red", "on_black", ["underline"]) + cprint("Reversed green on red color", "green", "on_red", ["reverse"]) + print("-" * 78) + + print("Test RGB:") + cprint("Pure red text (255, 0, 0)", (255, 0, 0)) + cprint("Default red for comparison", "red") + cprint("Pure green text (0, 0, 0)", (0, 255, 0)) + cprint("Default green for comparison", "green") + cprint("Pure blue text (0, 0, 0)", (0, 0, 255)) + cprint("Default blue for comparison", "blue") + cprint("Pure yellow text (255, 255, 0)", (255, 255, 0)) + cprint("Default yellow for comparison", "yellow") + cprint("Pure cyan text (0, 255, 255)", (0, 255, 255)) + cprint("Default cyan for comparison", "cyan") + cprint("Pure magenta text (255, 0, 255)", (255, 0, 255)) + cprint("Default magenta for comparison", "magenta") + cprint("Light pink (255, 182, 193)", (255, 182, 193)) + cprint("Light pink (255, 105, 180)", (255, 105, 180)) diff --git a/lib/python3.12/site-packages/termcolor/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/termcolor/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..57650a4e2dda66f7ac4c02e424b7c932b42d16dc Binary files /dev/null and b/lib/python3.12/site-packages/termcolor/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/termcolor/__pycache__/__main__.cpython-312.pyc b/lib/python3.12/site-packages/termcolor/__pycache__/__main__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..6b27668dda26024ec9fc52bbd33f3dc22f0afbcf Binary files /dev/null and b/lib/python3.12/site-packages/termcolor/__pycache__/__main__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/termcolor/__pycache__/termcolor.cpython-312.pyc b/lib/python3.12/site-packages/termcolor/__pycache__/termcolor.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..362f4a971999c10c9462a5a1c017da33eebc183e Binary files /dev/null and b/lib/python3.12/site-packages/termcolor/__pycache__/termcolor.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/termcolor/py.typed b/lib/python3.12/site-packages/termcolor/py.typed new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/lib/python3.12/site-packages/termcolor/termcolor.py b/lib/python3.12/site-packages/termcolor/termcolor.py new file mode 100644 index 0000000000000000000000000000000000000000..249158a0a8776bf6066424410997b3235124e625 --- /dev/null +++ b/lib/python3.12/site-packages/termcolor/termcolor.py @@ -0,0 +1,215 @@ +# Copyright (c) 2008-2011 Volvox Development Team +# +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in +# all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN +# THE SOFTWARE. +# +# Author: Konstantin Lepa + +"""ANSI color formatting for output in terminal.""" + +from __future__ import annotations + +import io +import os +import sys +from functools import cache + +TYPE_CHECKING = False +if TYPE_CHECKING: + from collections.abc import Iterable + from typing import Any + + +ATTRIBUTES: dict[str, int] = { + "bold": 1, + "dark": 2, + "underline": 4, + "blink": 5, + "reverse": 7, + "concealed": 8, + "strike": 9, +} + +HIGHLIGHTS: dict[str, int] = { + "on_black": 40, + "on_grey": 40, # Actually black but kept for backwards compatibility + "on_red": 41, + "on_green": 42, + "on_yellow": 43, + "on_blue": 44, + "on_magenta": 45, + "on_cyan": 46, + "on_light_grey": 47, + "on_dark_grey": 100, + "on_light_red": 101, + "on_light_green": 102, + "on_light_yellow": 103, + "on_light_blue": 104, + "on_light_magenta": 105, + "on_light_cyan": 106, + "on_white": 107, +} + +COLORS: dict[str, int] = { + "black": 30, + "grey": 30, # Actually black but kept for backwards compatibility + "red": 31, + "green": 32, + "yellow": 33, + "blue": 34, + "magenta": 35, + "cyan": 36, + "light_grey": 37, + "dark_grey": 90, + "light_red": 91, + "light_green": 92, + "light_yellow": 93, + "light_blue": 94, + "light_magenta": 95, + "light_cyan": 96, + "white": 97, +} + + +RESET = "\033[0m" + + +@cache +def _can_do_colour( + *, no_color: bool | None = None, force_color: bool | None = None +) -> bool: + """Check env vars and for tty/dumb terminal""" + # First check overrides: + # "User-level configuration files and per-instance command-line arguments should + # override $NO_COLOR. A user should be able to export $NO_COLOR in their shell + # configuration file as a default, but configure a specific program in its + # configuration file to specifically enable color." + # https://no-color.org + if no_color is not None and no_color: + return False + if force_color is not None and force_color: + return True + + # Then check env vars: + if os.environ.get("ANSI_COLORS_DISABLED"): + return False + if os.environ.get("NO_COLOR"): + return False + if os.environ.get("FORCE_COLOR"): + return True + + # Then check system: + if os.environ.get("TERM") == "dumb": + return False + if not hasattr(sys.stdout, "fileno"): + return False + + try: + return os.isatty(sys.stdout.fileno()) + except io.UnsupportedOperation: + return sys.stdout.isatty() + + +def colored( + text: object, + color: str | tuple[int, int, int] | None = None, + on_color: str | tuple[int, int, int] | None = None, + attrs: Iterable[str] | None = None, + *, + no_color: bool | None = None, + force_color: bool | None = None, +) -> str: + """Colorize text. + + Available text colors: + black, red, green, yellow, blue, magenta, cyan, white, + light_grey, dark_grey, light_red, light_green, light_yellow, light_blue, + light_magenta, light_cyan. + + Available text highlights: + on_black, on_red, on_green, on_yellow, on_blue, on_magenta, on_cyan, on_white, + on_light_grey, on_dark_grey, on_light_red, on_light_green, on_light_yellow, + on_light_blue, on_light_magenta, on_light_cyan. + + Alternatively, both text colors (color) and highlights (on_color) may + be specified via a tuple of 0-255 ints (R, G, B). + + Available attributes: + bold, dark, underline, blink, reverse, concealed. + + Example: + colored('Hello, World!', 'red', 'on_black', ['bold', 'blink']) + colored('Hello, World!', 'green') + colored('Hello, World!', (255, 0, 255)) # Purple + """ + result = str(text) + if not _can_do_colour(no_color=no_color, force_color=force_color): + return result + + fmt_str = "\033[%dm%s" + rgb_fore_fmt_str = "\033[38;2;%d;%d;%dm%s" + rgb_back_fmt_str = "\033[48;2;%d;%d;%dm%s" + if color is not None: + if isinstance(color, str): + result = fmt_str % (COLORS[color], result) + elif isinstance(color, tuple): + result = rgb_fore_fmt_str % (color[0], color[1], color[2], result) + + if on_color is not None: + if isinstance(on_color, str): + result = fmt_str % (HIGHLIGHTS[on_color], result) + elif isinstance(on_color, tuple): + result = rgb_back_fmt_str % (on_color[0], on_color[1], on_color[2], result) + + if attrs is not None: + for attr in attrs: + result = fmt_str % (ATTRIBUTES[attr], result) + + result += RESET + + return result + + +def cprint( + text: object, + color: str | tuple[int, int, int] | None = None, + on_color: str | tuple[int, int, int] | None = None, + attrs: Iterable[str] | None = None, + *, + no_color: bool | None = None, + force_color: bool | None = None, + **kwargs: Any, +) -> None: + """Print colorized text. + + It accepts arguments of print function. + """ + + print( + ( + colored( + text, + color, + on_color, + attrs, + no_color=no_color, + force_color=force_color, + ) + ), + **kwargs, + ) diff --git a/lib/python3.12/site-packages/wandb/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e47a3fc31181c47b144f1ab1387b03a24167bccc Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/__pycache__/__main__.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/__main__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7be44c74c483709e3ec5a824a877cd93215fe4fa Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/__main__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/__pycache__/_iterutils.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/_iterutils.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..924e50f978cad05ae9d07e7902cf0b542522574d Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/_iterutils.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/__pycache__/data_types.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/data_types.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c8baf08297a4db24514f681af248113e6098f7fc Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/data_types.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/__pycache__/env.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/env.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b039943cd1384b2665e24f992685a8ba808c0f7b Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/env.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/__pycache__/jupyter.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/jupyter.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..8d1ebbdafa6be8d048e2b50ebabd894bb1511001 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/jupyter.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/__pycache__/sklearn.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/sklearn.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..3c3ba060216c1ad1c16705bfe988880691638256 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/sklearn.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/__pycache__/trigger.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/trigger.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..46982a2957d8177d6c7b9abc1f6e010fe933a0f7 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/trigger.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/__pycache__/util.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/util.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..09c1419cd800ae340ff631d966b0c4c64e077511 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/util.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/__pycache__/wandb_agent.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/wandb_agent.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c8f60e3f7bd0387bb2fb87d3a081bee62956e60b Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/wandb_agent.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/__pycache__/wandb_controller.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/wandb_controller.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ad7b7d767a52b493065aaea4cc34251cdff4ac76 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/wandb_controller.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/__pycache__/wandb_run.cpython-312.pyc b/lib/python3.12/site-packages/wandb/__pycache__/wandb_run.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a88ae5461e80418bd2e9b7a4e9e0584cdea36a33 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/__pycache__/wandb_run.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/cli/__init__.py b/lib/python3.12/site-packages/wandb/cli/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/lib/python3.12/site-packages/wandb/cli/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/wandb/cli/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a04c61aa2374ddd8d53543fed748586c210f99e4 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/cli/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/cli/__pycache__/beta.cpython-312.pyc b/lib/python3.12/site-packages/wandb/cli/__pycache__/beta.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..6082dec3308afb813dcdab7c9c72b9f7d7457cee Binary files /dev/null and b/lib/python3.12/site-packages/wandb/cli/__pycache__/beta.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/cli/beta.py b/lib/python3.12/site-packages/wandb/cli/beta.py new file mode 100644 index 0000000000000000000000000000000000000000..324daf0c5e3196bb87cd060b6da4e15bab9ad087 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/cli/beta.py @@ -0,0 +1,175 @@ +"""Beta versions of wandb CLI commands. + +These commands are experimental and may change or be removed in future versions. +""" + +from __future__ import annotations + +import pathlib +import sys + +import click + +import wandb +from wandb.errors import WandbCoreNotAvailableError +from wandb.sdk.wandb_sync import _sync +from wandb.util import get_core_path + + +@click.group() +def beta(): + """Beta versions of wandb CLI commands. Requires wandb-core.""" + # this is the future that requires wandb-core! + import wandb.env + + wandb._sentry.configure_scope(process_context="wandb_beta") + + try: + get_core_path() + except WandbCoreNotAvailableError as e: + wandb._sentry.exception(f"using `wandb beta`. failed with {e}") + click.secho( + (e), + fg="red", + err=True, + ) + + +@beta.command( + name="sync", + context_settings={"default_map": {}}, + help="Upload a training run to W&B", +) +@click.pass_context +@click.argument("wandb_dir", nargs=1, type=click.Path(exists=True)) +@click.option("--id", "run_id", help="The run you want to upload to.") +@click.option("--project", "-p", help="The project you want to upload to.") +@click.option("--entity", "-e", help="The entity to scope to.") +@click.option("--skip-console", is_flag=True, default=False, help="Skip console logs") +@click.option("--append", is_flag=True, default=False, help="Append run") +@click.option( + "--include", + "-i", + help="Glob to include. Can be used multiple times.", + multiple=True, +) +@click.option( + "--exclude", + "-e", + help="Glob to exclude. Can be used multiple times.", + multiple=True, +) +@click.option( + "--mark-synced/--no-mark-synced", + is_flag=True, + default=True, + help="Mark runs as synced", +) +@click.option( + "--skip-synced/--no-skip-synced", + is_flag=True, + default=True, + help="Skip synced runs", +) +@click.option( + "--dry-run", is_flag=True, help="Perform a dry run without uploading anything." +) +def sync_beta( # noqa: C901 + ctx, + wandb_dir=None, + run_id: str | None = None, + project: str | None = None, + entity: str | None = None, + skip_console: bool = False, + append: bool = False, + include: str | None = None, + exclude: str | None = None, + skip_synced: bool = True, + mark_synced: bool = True, + dry_run: bool = False, +) -> None: + import concurrent.futures + from multiprocessing import cpu_count + + paths = set() + + # TODO: test file discovery logic + # include and exclude globs are evaluated relative to the provided base_path + if include: + for pattern in include: + matching_dirs = list(pathlib.Path(wandb_dir).glob(pattern)) + for d in matching_dirs: + if not d.is_dir(): + continue + wandb_files = [p for p in d.glob("*.wandb") if p.is_file()] + if len(wandb_files) > 1: + wandb.termwarn( + f"Multiple wandb files found in directory {d}, skipping" + ) + elif len(wandb_files) == 1: + paths.add(d) + else: + paths.update({p.parent for p in pathlib.Path(wandb_dir).glob("**/*.wandb")}) + + for pattern in exclude: + matching_dirs = list(pathlib.Path(wandb_dir).glob(pattern)) + for d in matching_dirs: + if not d.is_dir(): + continue + if d in paths: + paths.remove(d) + + # remove paths that are already synced, if requested + if skip_synced: + synced_paths = set() + for path in paths: + wandb_synced_files = [p for p in path.glob("*.wandb.synced") if p.is_file()] + if len(wandb_synced_files) > 1: + wandb.termwarn( + f"Multiple wandb.synced files found in directory {path}, skipping" + ) + elif len(wandb_synced_files) == 1: + synced_paths.add(path) + paths -= synced_paths + + if run_id and len(paths) > 1: + # TODO: handle this more gracefully + click.echo("id can only be set for a single run.", err=True) + sys.exit(1) + + if not paths: + click.echo("No runs to sync.") + return + + click.echo("Found runs:") + for path in paths: + click.echo(f" {path}") + + if dry_run: + return + + wandb.setup() + + # TODO: make it thread-safe in the Rust code + with concurrent.futures.ProcessPoolExecutor( + max_workers=min(len(paths), cpu_count()) + ) as executor: + futures = [] + for path in paths: + # we already know there is only one wandb file in the directory + wandb_file = [p for p in path.glob("*.wandb") if p.is_file()][0] + future = executor.submit( + _sync, + wandb_file, + run_id=run_id, + project=project, + entity=entity, + skip_console=skip_console, + append=append, + mark_synced=mark_synced, + ) + futures.append(future) + + # Wait for tasks to complete + for _ in concurrent.futures.as_completed(futures): + pass diff --git a/lib/python3.12/site-packages/wandb/cli/cli.py b/lib/python3.12/site-packages/wandb/cli/cli.py new file mode 100644 index 0000000000000000000000000000000000000000..3f26a2729cdf7faedcd97b1d66846c8042c10688 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/cli/cli.py @@ -0,0 +1,2759 @@ +import asyncio +import configparser +import datetime +import getpass +import json +import logging +import os +import shlex +import shutil +import subprocess +import sys +import tempfile +import textwrap +import time +import traceback +from functools import wraps +from typing import Any, Dict, Optional + +import click +import yaml +from click.exceptions import ClickException + +import wandb +import wandb.env +import wandb.errors +import wandb.sdk.verify.verify as wandb_verify +from wandb import Config, Error, env, util, wandb_agent, wandb_sdk +from wandb.apis import InternalApi, PublicApi +from wandb.apis.public import RunQueue +from wandb.errors.links import url_registry +from wandb.sdk import wandb_setup +from wandb.sdk.artifacts._validators import is_artifact_registry_project +from wandb.sdk.artifacts.artifact_file_cache import get_artifact_file_cache +from wandb.sdk.internal.internal_api import Api as SDKInternalApi +from wandb.sdk.launch import utils as launch_utils +from wandb.sdk.launch._launch_add import _launch_add +from wandb.sdk.launch.errors import ExecutionError, LaunchError +from wandb.sdk.launch.sweeps import utils as sweep_utils +from wandb.sdk.launch.sweeps.scheduler import Scheduler +from wandb.sdk.lib import filesystem +from wandb.sync import SyncManager, get_run_from_path, get_runs + +from .beta import beta + +# Send cli logs to wandb/debug-cli..log by default and fallback to a temp dir. +_wandb_dir = wandb.old.core.wandb_dir(env.get_dir()) +if not os.path.exists(_wandb_dir): + _wandb_dir = tempfile.gettempdir() + +try: + _username = getpass.getuser() +except KeyError: + # getuser() could raise KeyError in restricted environments like + # chroot jails or docker containers. Return user id in these cases. + _username = str(os.getuid()) + +_wandb_log_path = os.path.join(_wandb_dir, f"debug-cli.{_username}.log") + +logging.basicConfig( + filename=_wandb_log_path, + level=logging.INFO, + format="%(asctime)s %(levelname)s %(message)s", + datefmt="%Y-%m-%d %H:%M:%S", +) +logging.basicConfig(stream=sys.stdout, level=logging.INFO) +logger = logging.getLogger("wandb") + +_HAS_DOCKER = bool(shutil.which("docker")) +_HAS_NVIDIA_DOCKER = bool(shutil.which("nvidia-docker")) + +# Click Contexts +CONTEXT = {"default_map": {}} +RUN_CONTEXT = { + "default_map": {}, + "allow_extra_args": True, + "ignore_unknown_options": True, +} + + +def cli_unsupported(argument): + wandb.termerror(f"Unsupported argument `{argument}`") + sys.exit(1) + + +class ClickWandbException(ClickException): + def format_message(self): + orig_type = f"{self.orig_type.__module__}.{self.orig_type.__name__}" + if issubclass(self.orig_type, Error): + return click.style(str(self.message), fg="red") + else: + return ( + f"An Exception was raised, see {_wandb_log_path} for full" + " traceback.\n" + f"{orig_type}: {self.message}" + ) + + +def display_error(func): + """Function decorator for catching common errors and re-raising as wandb.Error.""" + + @wraps(func) + def wrapper(*args, **kwargs): + try: + return func(*args, **kwargs) + except wandb.Error as e: + exc_type, exc_value, exc_traceback = sys.exc_info() + lines = traceback.format_exception(exc_type, exc_value, exc_traceback) + logger.exception("".join(lines)) + wandb.termerror(f"Find detailed error logs at: {_wandb_log_path}") + click_exc = ClickWandbException(e) + click_exc.orig_type = exc_type + raise click_exc.with_traceback(sys.exc_info()[2]) + + return wrapper + + +_api = None # caching api instance allows patching from unit tests + + +def _get_cling_api(reset=None): + """Get a reference to the internal api with cling settings.""" + global _api + if reset: + _api = None + wandb.teardown() + if _api is None: + # TODO(jhr): make a settings object that is better for non runs. + # only override the necessary setting + wandb_setup.singleton().settings.x_cli_only_mode = True + _api = InternalApi() + return _api + + +def prompt_for_project(ctx, entity): + """Ask the user for a project, creating one if necessary.""" + result = ctx.invoke(projects, entity=entity, display=False) + api = _get_cling_api() + try: + if len(result) == 0: + project = click.prompt("Enter a name for your first project") + # description = editor() + project = api.upsert_project(project, entity=entity)["name"] + else: + project_names = [project["name"] for project in result] + ["Create New"] + wandb.termlog("Which project should we use?") + result = util.prompt_choices(project_names) + if result: + project = result + else: + project = "Create New" + # TODO: check with the server if the project exists + if project == "Create New": + project = click.prompt( + "Enter a name for your new project", value_proc=api.format_project + ) + # description = editor() + project = api.upsert_project(project, entity=entity)["name"] + + except wandb.errors.CommError as e: + raise ClickException(str(e)) + + return project + + +class RunGroup(click.Group): + @display_error + def get_command(self, ctx, cmd_name): + # TODO: check if cmd_name is a file in the current dir and not require `run`? + rv = click.Group.get_command(self, ctx, cmd_name) + if rv is not None: + return rv + return None + + +@click.command(cls=RunGroup, invoke_without_command=True) +@click.version_option(version=wandb.__version__) +@click.pass_context +def cli(ctx): + if ctx.invoked_subcommand is None: + click.echo(ctx.get_help()) + + +@cli.command(context_settings=CONTEXT, help="List projects", hidden=True) +@click.option( + "--entity", + "-e", + default=None, + envvar=env.ENTITY, + help="The entity to scope the listing to.", +) +@display_error +def projects(entity, display=True): + api = _get_cling_api() + projects = api.list_projects(entity=entity) + if len(projects) == 0: + message = f"No projects found for {entity}" + else: + message = f'Latest projects for "{entity}"' + if display: + click.echo(click.style(message, bold=True)) + for project in projects: + click.echo( + "".join( + ( + click.style(project["name"], fg="blue", bold=True), + " - ", + str(project["description"] or "").split("\n")[0], + ) + ) + ) + return projects + + +@cli.command(context_settings=CONTEXT, help="Login to Weights & Biases") +@click.argument("key", nargs=-1) +@click.option("--cloud", is_flag=True, help="Login to the cloud instead of local") +@click.option( + "--host", "--base-url", default=None, help="Login to a specific instance of W&B" +) +@click.option( + "--relogin", default=None, is_flag=True, help="Force relogin if already logged in." +) +@click.option("--anonymously", default=False, is_flag=True, help="Log in anonymously") +@click.option( + "--verify/--no-verify", + default=False, + is_flag=True, + help="Verify login credentials", +) +@display_error +def login(key, host, cloud, relogin, anonymously, verify, no_offline=False): + # TODO: handle no_offline + anon_mode = "must" if anonymously else "never" + + wandb_sdk.wandb_login._handle_host_wandb_setting(host, cloud) + # A change in click or the test harness means key can be none... + key = key[0] if key is not None and len(key) > 0 else None + relogin = True if key or relogin else False + + global_settings = wandb_setup.singleton().settings + global_settings.x_cli_only_mode = True + global_settings.x_disable_viewer = relogin and not verify + + wandb.login( + anonymous=anon_mode, + force=True, + host=host, + key=key, + relogin=relogin, + verify=verify, + referrer="models", + ) + + +@cli.command( + context_settings=CONTEXT, help="Configure a directory with Weights & Biases" +) +@click.option("--project", "-p", help="The project to use.") +@click.option("--entity", "-e", help="The entity to scope the project to.") +# TODO(jhr): Enable these with settings rework +# @click.option("--setting", "-s", help="enable an arbitrary setting.", multiple=True) +# @click.option('--show', is_flag=True, help="Show settings") +@click.option("--reset", is_flag=True, help="Reset settings") +@click.option( + "--mode", + "-m", + help=' Can be "online", "offline" or "disabled". Defaults to online.', +) +@click.pass_context +@display_error +def init(ctx, project, entity, reset, mode): + from wandb.old.core import __stage_dir__, _set_stage_dir, wandb_dir + + if __stage_dir__ is None: + _set_stage_dir("wandb") + + # non-interactive init + if reset or project or entity or mode: + api = InternalApi() + if reset: + api.clear_setting("entity", persist=True) + api.clear_setting("project", persist=True) + api.clear_setting("mode", persist=True) + # TODO(jhr): clear more settings? + if entity: + api.set_setting("entity", entity, persist=True) + if project: + api.set_setting("project", project, persist=True) + if mode: + api.set_setting("mode", mode, persist=True) + return + + if os.path.isdir(wandb_dir()) and os.path.exists( + os.path.join(wandb_dir(), "settings") + ): + click.confirm( + click.style( + "This directory has been configured previously, should we re-configure it?", + bold=True, + ), + abort=True, + ) + else: + click.echo( + click.style("Let's setup this directory for W&B!", fg="green", bold=True) + ) + api = _get_cling_api() + if api.api_key is None: + ctx.invoke(login) + api = _get_cling_api(reset=True) + + viewer = api.viewer() + + # Viewer can be `None` in case your API information became invalid, or + # in testing if you switch hosts. + if not viewer: + click.echo( + click.style( + "Your login information seems to be invalid: can you log in again please?", + fg="red", + bold=True, + ) + ) + ctx.invoke(login) + api = _get_cling_api(reset=True) + + # This shouldn't happen. + viewer = api.viewer() + if not viewer: + click.echo( + click.style( + "We're sorry, there was a problem logging you in. " + "Please send us a note at support@wandb.com and tell us how this happened.", + fg="red", + bold=True, + ) + ) + sys.exit(1) + + # At this point we should be logged in successfully. + if len(viewer["teams"]["edges"]) > 1: + team_names = [e["node"]["name"] for e in viewer["teams"]["edges"]] + [ + "Manual entry" + ] + wandb.termlog( + "Which team should we use?", + ) + result = util.prompt_choices(team_names) + # result can be empty on click + if result: + entity = result + else: + entity = "Manual Entry" + if entity == "Manual Entry": + entity = click.prompt("Enter the name of the team you want to use") + else: + entity = viewer.get("entity") or click.prompt( + "What username or team should we use?" + ) + + # TODO: this error handling sucks and the output isn't pretty + try: + project = prompt_for_project(ctx, entity) + except ClickWandbException: + raise ClickException(f"Could not find team: {entity}") + + api.set_setting("entity", entity, persist=True) + api.set_setting("project", project, persist=True) + api.set_setting("base_url", api.settings().get("base_url"), persist=True) + + filesystem.mkdir_exists_ok(wandb_dir()) + with open(os.path.join(wandb_dir(), ".gitignore"), "w") as file: + file.write("*\n!settings") + + click.echo( + click.style("This directory is configured! Next, track a run:\n", fg="green") + + textwrap.dedent( + """\ + * In your training script: + {code1} + {code2} + * then `{run}`. + """ + ).format( + code1=click.style("import wandb", bold=True), + code2=click.style(f'wandb.init(project="{project}")', bold=True), + run=click.style("python ", bold=True), + ) + ) + + +@cli.command( + context_settings=CONTEXT, help="Upload an offline training directory to W&B" +) +@click.pass_context +@click.argument("path", nargs=-1, type=click.Path(exists=True)) +@click.option("--view", is_flag=True, default=False, help="View runs", hidden=True) +@click.option("--verbose", is_flag=True, default=False, help="Verbose", hidden=True) +@click.option("--id", "run_id", help="The run you want to upload to.") +@click.option("--project", "-p", help="The project you want to upload to.") +@click.option("--entity", "-e", help="The entity to scope to.") +@click.option( + "--job_type", + "job_type", + help="Specifies the type of run for grouping related runs together.", +) +@click.option( + "--sync-tensorboard/--no-sync-tensorboard", + is_flag=True, + default=None, + help="Stream tfevent files to wandb.", +) +@click.option("--include-globs", help="Comma separated list of globs to include.") +@click.option("--exclude-globs", help="Comma separated list of globs to exclude.") +@click.option( + "--include-online/--no-include-online", + is_flag=True, + default=None, + help="Include online runs", +) +@click.option( + "--include-offline/--no-include-offline", + is_flag=True, + default=None, + help="Include offline runs", +) +@click.option( + "--include-synced/--no-include-synced", + is_flag=True, + default=None, + help="Include synced runs", +) +@click.option( + "--mark-synced/--no-mark-synced", + is_flag=True, + default=True, + help="Mark runs as synced", +) +@click.option("--sync-all", is_flag=True, default=False, help="Sync all runs") +@click.option("--clean", is_flag=True, default=False, help="Delete synced runs") +@click.option( + "--clean-old-hours", + default=24, + help="Delete runs created before this many hours. To be used alongside --clean flag.", + type=int, +) +@click.option( + "--clean-force", + is_flag=True, + default=False, + help="Clean without confirmation prompt.", +) +@click.option("--ignore", hidden=True) +@click.option("--show", default=5, help="Number of runs to show") +@click.option("--append", is_flag=True, default=False, help="Append run") +@click.option("--skip-console", is_flag=True, default=False, help="Skip console logs") +@display_error +def sync( + ctx, + path=None, + view=None, + verbose=None, + run_id=None, + project=None, + entity=None, + job_type=None, # trace this back to SyncManager + sync_tensorboard=None, + include_globs=None, + exclude_globs=None, + include_online=None, + include_offline=None, + include_synced=None, + mark_synced=None, + sync_all=None, + ignore=None, + show=None, + clean=None, + clean_old_hours=24, + clean_force=None, + append=None, + skip_console=None, +): + api = _get_cling_api() + if not api.is_authenticated: + wandb.termlog("Login to W&B to sync offline runs") + ctx.invoke(login, no_offline=True) + api = _get_cling_api(reset=True) + + if ignore: + exclude_globs = ignore + if include_globs: + include_globs = include_globs.split(",") + if exclude_globs: + exclude_globs = exclude_globs.split(",") + + def _summary(): + all_items = get_runs( + include_online=True, + include_offline=True, + include_synced=True, + include_unsynced=True, + ) + sync_items = get_runs( + include_online=include_online if include_online is not None else True, + include_offline=include_offline if include_offline is not None else True, + include_synced=include_synced if include_synced is not None else False, + include_unsynced=True, + exclude_globs=exclude_globs, + include_globs=include_globs, + ) + synced = [] + unsynced = [] + for item in all_items: + (synced if item.synced else unsynced).append(item) + if sync_items: + wandb.termlog(f"Number of runs to be synced: {len(sync_items)}") + if show and show < len(sync_items): + wandb.termlog(f"Showing {show} runs to be synced:") + for item in sync_items[: (show or len(sync_items))]: + wandb.termlog(f" {item}") + else: + wandb.termlog("No runs to be synced.") + if synced: + clean_cmd = click.style("wandb sync --clean", fg="yellow") + wandb.termlog( + f"NOTE: use {clean_cmd} to delete {len(synced)} synced runs from local directory." + ) + if unsynced: + sync_cmd = click.style("wandb sync --sync-all", fg="yellow") + wandb.termlog( + f"NOTE: use {sync_cmd} to sync {len(unsynced)} unsynced runs from local directory." + ) + + def _sync_path(_path, _sync_tensorboard): + if run_id and len(_path) > 1: + wandb.termerror("id can only be set for a single run.") + sys.exit(1) + sm = SyncManager( + project=project, + entity=entity, + run_id=run_id, + job_type=job_type, + mark_synced=mark_synced, + app_url=api.app_url, + view=view, + verbose=verbose, + sync_tensorboard=_sync_tensorboard, + log_path=_wandb_log_path, + append=append, + skip_console=skip_console, + ) + for p in _path: + sm.add(p) + sm.start() + while not sm.is_done(): + _ = sm.poll() + + def _sync_all(): + sync_items = get_runs( + include_online=include_online if include_online is not None else True, + include_offline=include_offline if include_offline is not None else True, + include_synced=include_synced if include_synced is not None else False, + include_unsynced=True, + exclude_globs=exclude_globs, + include_globs=include_globs, + ) + if not sync_items: + wandb.termerror("Nothing to sync.") + else: + # When syncing run directories, default to not syncing tensorboard + sync_tb = sync_tensorboard if sync_tensorboard is not None else False + _sync_path(sync_items, sync_tb) + + def _clean(): + if path: + runs = list(map(get_run_from_path, path)) + if not clean_force: + click.confirm( + click.style( + f"Are you sure you want to remove {len(runs)} runs?", + bold=True, + ), + abort=True, + ) + for run in runs: + shutil.rmtree(run.path) + click.echo(click.style("Success!", fg="green")) + return + runs = get_runs( + include_online=include_online if include_online is not None else True, + include_offline=include_offline if include_offline is not None else True, + include_synced=include_synced if include_synced is not None else True, + include_unsynced=False, + exclude_globs=exclude_globs, + include_globs=include_globs, + ) + since = datetime.datetime.now() - datetime.timedelta(hours=clean_old_hours) + old_runs = [run for run in runs if run.datetime < since] + old_runs.sort(key=lambda _run: _run.datetime) + if old_runs: + click.echo( + f"Found {len(runs)} runs, {len(old_runs)} are older than {clean_old_hours} hours" + ) + for run in old_runs: + click.echo(run.path) + if not clean_force: + click.confirm( + click.style( + f"Are you sure you want to remove {len(old_runs)} runs?", + bold=True, + ), + abort=True, + ) + for run in old_runs: + shutil.rmtree(run.path) + click.echo(click.style("Success!", fg="green")) + else: + click.echo( + click.style( + f"No runs older than {clean_old_hours} hours found", fg="red" + ) + ) + + if sync_all: + _sync_all() + elif clean: + _clean() + elif path: + # When syncing a specific path, default to syncing tensorboard + sync_tb = sync_tensorboard if sync_tensorboard is not None else True + _sync_path(path, sync_tb) + else: + _summary() + + +@cli.command( + context_settings=CONTEXT, + help="Initialize a hyperparameter sweep. Search for hyperparameters that optimizes a cost function of a machine learning model by testing various combinations.", +) +@click.option( + "--project", + "-p", + default=None, + help="""The name of the project where W&B runs created from the sweep are sent to. If the project is not specified, the run is sent to a project labeled Uncategorized.""", +) +@click.option( + "--entity", + "-e", + default=None, + help="""The username or team name where you want to send W&B runs created by the sweep to. Ensure that the entity you specify already exists. If you don't specify an entity, the run will be sent to your default entity, which is usually your username.""", +) +@click.option("--controller", is_flag=True, default=False, help="Run local controller") +@click.option("--verbose", is_flag=True, default=False, help="Display verbose output") +@click.option( + "--name", + default=None, + help="The name of the sweep. The sweep ID is used if no name is specified.", +) +@click.option("--program", default=None, help="Set sweep program") +@click.option("--settings", default=None, help="Set sweep settings", hidden=True) +@click.option("--update", default=None, help="Update pending sweep") +@click.option( + "--stop", + is_flag=True, + default=False, + help="Finish a sweep to stop running new runs and let currently running runs finish.", +) +@click.option( + "--cancel", + is_flag=True, + default=False, + help="Cancel a sweep to kill all running runs and stop running new runs.", +) +@click.option( + "--pause", + is_flag=True, + default=False, + help="Pause a sweep to temporarily stop running new runs.", +) +@click.option( + "--resume", + is_flag=True, + default=False, + help="Resume a sweep to continue running new runs.", +) +@click.option( + "--prior_run", + "-R", + "prior_runs", + multiple=True, + default=None, + help="ID of an existing run to add to this sweep", +) +@click.argument("config_yaml_or_sweep_id") +@click.pass_context +@display_error +def sweep( + ctx, + project, + entity, + controller, + verbose, + name, + program, + settings, + update, + stop, + cancel, + pause, + resume, + prior_runs, + config_yaml_or_sweep_id, +): + state_args = "stop", "cancel", "pause", "resume" + lcls = locals() + is_state_change_command = sum(lcls[k] for k in state_args) + if is_state_change_command > 1: + raise Exception("Only one state flag (stop/cancel/pause/resume) is allowed.") + elif is_state_change_command == 1: + sweep_id = config_yaml_or_sweep_id + api = _get_cling_api() + if not api.is_authenticated: + wandb.termlog("Login to W&B to use the sweep feature") + ctx.invoke(login, no_offline=True) + api = _get_cling_api(reset=True) + parts = dict(entity=entity, project=project, name=sweep_id) + err = sweep_utils.parse_sweep_id(parts) + if err: + wandb.termerror(err) + return + entity = parts.get("entity") or entity + project = parts.get("project") or project + sweep_id = parts.get("name") or sweep_id + state = [s for s in state_args if lcls[s]][0] + ings = { + "stop": "Stopping", + "cancel": "Cancelling", + "pause": "Pausing", + "resume": "Resuming", + } + wandb.termlog(f"{ings[state]} sweep {entity}/{project}/{sweep_id}") + getattr(api, f"{state}_sweep")(sweep_id, entity=entity, project=project) + wandb.termlog("Done.") + return + else: + config_yaml = config_yaml_or_sweep_id + + def _parse_settings(settings): + """Parse settings from json or comma separated assignments.""" + ret = {} + # TODO(jhr): merge with magic:_parse_magic + if settings.find("=") > 0: + for item in settings.split(","): + kv = item.split("=") + if len(kv) != 2: + wandb.termwarn( + "Unable to parse sweep settings key value pair", repeat=False + ) + ret.update(dict([kv])) + return ret + wandb.termwarn("Unable to parse settings parameter", repeat=False) + return ret + + api = _get_cling_api() + if not api.is_authenticated: + wandb.termlog("Login to W&B to use the sweep feature") + ctx.invoke(login, no_offline=True) + api = _get_cling_api(reset=True) + + sweep_obj_id = None + if update: + parts = dict(entity=entity, project=project, name=update) + err = sweep_utils.parse_sweep_id(parts) + if err: + wandb.termerror(err) + return + entity = parts.get("entity") or entity + project = parts.get("project") or project + sweep_id = parts.get("name") or update + + has_project = (project or api.settings("project")) is not None + has_entity = (entity or api.settings("entity")) is not None + + termerror_msg = ( + "Sweep lookup requires a valid %s, and none was specified. \n" + "Either set a default %s in wandb/settings, or, if invoking \n`wandb sweep` " + "from the command line, specify the full sweep path via: \n\n" + " wandb sweep {username}/{projectname}/{sweepid}\n\n" + ) + + if not has_entity: + wandb.termerror(termerror_msg % (("entity",) * 2)) + return + + if not has_project: + wandb.termerror(termerror_msg % (("project",) * 2)) + return + + found = api.sweep(sweep_id, "{}", entity=entity, project=project) + if not found: + wandb.termerror(f"Could not find sweep {entity}/{project}/{sweep_id}") + return + sweep_obj_id = found["id"] + + action = "Updating" if sweep_obj_id else "Creating" + wandb.termlog(f"{action} sweep from: {config_yaml}") + config = sweep_utils.load_sweep_config(config_yaml) + + # Set or override parameters + if name: + config["name"] = name + if program: + config["program"] = program + if settings: + settings = _parse_settings(settings) + if settings: + config.setdefault("settings", {}) + config["settings"].update(settings) + if controller: + config.setdefault("controller", {}) + config["controller"]["type"] = "local" + + is_local = config.get("controller", {}).get("type") == "local" + if is_local: + from wandb import controller as wandb_controller + + tuner = wandb_controller() + err = tuner._validate(config) + if err: + wandb.termerror(f"Error in sweep file: {err}") + return + + env = os.environ + entity = ( + entity + or env.get("WANDB_ENTITY") + or config.get("entity") + or api.settings("entity") + ) + project = ( + project + or env.get("WANDB_PROJECT") + or config.get("project") + or api.settings("project") + or util.auto_project_name(config.get("program")) + ) + + sweep_id, warnings = api.upsert_sweep( + config, + project=project, + entity=entity, + obj_id=sweep_obj_id, + prior_runs=prior_runs, + ) + sweep_utils.handle_sweep_config_violations(warnings) + + # Log nicely formatted sweep information + styled_id = click.style(sweep_id, fg="yellow") + wandb.termlog(f"{action} sweep with ID: {styled_id}") + + sweep_url = wandb_sdk.wandb_sweep._get_sweep_url(api, sweep_id) + if sweep_url: + styled_url = click.style(sweep_url, underline=True, fg="blue") + wandb.termlog(f"View sweep at: {styled_url}") + + # re-probe entity and project if it was auto-detected by upsert_sweep + entity = entity or env.get("WANDB_ENTITY") + project = project or env.get("WANDB_PROJECT") + + if entity and project: + sweep_path = f"{entity}/{project}/{sweep_id}" + elif project: + sweep_path = f"{project}/{sweep_id}" + else: + sweep_path = sweep_id + + if sweep_path.find(" ") >= 0: + sweep_path = f"{sweep_path!r}" + + styled_path = click.style(f"wandb agent {sweep_path}", fg="yellow") + wandb.termlog(f"Run sweep agent with: {styled_path}") + if controller: + wandb.termlog("Starting wandb controller...") + from wandb import controller as wandb_controller + + tuner = wandb_controller(sweep_id) + tuner.run(verbose=verbose) + + +@cli.command( + context_settings=CONTEXT, + no_args_is_help=True, + help="Run a W&B launch sweep (Experimental).", +) +@click.option( + "--queue", + "-q", + default=None, + help="The name of a queue to push the sweep to", +) +@click.option( + "--project", + "-p", + default=None, + help="Name of the project which the agent will watch. " + "If passed in, will override the project value passed in using a config file", +) +@click.option( + "--entity", + "-e", + default=None, + help="The entity to use. Defaults to current logged-in user", +) +@click.option( + "--resume_id", + "-r", + default=None, + help="Resume a launch sweep by passing an 8-char sweep id. Queue required", +) +@click.option( + "--prior_run", + "-R", + "prior_runs", + multiple=True, + default=None, + help="ID of an existing run to add to this sweep", +) +@click.argument("config", required=False, type=click.Path(exists=True)) +@click.pass_context +@display_error +def launch_sweep( + ctx, + project, + entity, + queue, + config, + resume_id, + prior_runs, +): + api = _get_cling_api() + env = os.environ + if not api.is_authenticated: + wandb.termlog("Login to W&B to use the sweep feature") + ctx.invoke(login, no_offline=True) + api = _get_cling_api(reset=True) + + entity = entity or env.get("WANDB_ENTITY") or api.settings("entity") + if entity is None: + wandb.termerror("Must specify entity when using launch") + return + + project = project or env.get("WANDB_PROJECT") or api.settings("project") + if project is None: + wandb.termerror("A project must be configured when using launch") + return + + # get personal username, not team name or service account, default to entity + author = api.viewer().get("username") or entity + + # if not sweep_config XOR resume_id + if not (config or resume_id): + wandb.termerror("'config' and/or 'resume_id' required") + return + + parsed_user_config = sweep_utils.load_launch_sweep_config(config) + # Rip special keys out of config, store in scheduler run_config + launch_args: Dict[str, Any] = parsed_user_config.pop("launch", {}) + scheduler_args: Dict[str, Any] = parsed_user_config.pop("scheduler", {}) + settings: Dict[str, Any] = scheduler_args.pop("settings", {}) + + scheduler_job: Optional[str] = scheduler_args.get("job") + if scheduler_job: + wandb.termwarn( + "Using a scheduler job for launch sweeps is *experimental* and may change without warning" + ) + queue: Optional[str] = queue or launch_args.get("queue") + + sweep_config, sweep_obj_id = None, None + if not resume_id: + sweep_config = parsed_user_config + + # check method + method = sweep_config.get("method") + if scheduler_job and not method: + sweep_config["method"] = "custom" + elif scheduler_job and method != "custom": + # TODO(gst): Check if using Anaconda2 + wandb.termwarn( + "Use 'method': 'custom' in the sweep config when using scheduler jobs, " + "or omit it entirely. For jobs using the wandb optimization engine (WandbScheduler), " + "set the method in the sweep config under scheduler.settings.method " + ) + settings["method"] = method + + if settings.get("method"): + # assume WandbScheduler, and user is using this right + sweep_config["method"] = settings["method"] + + else: # Resuming an existing sweep + found = api.sweep(resume_id, "{}", entity=entity, project=project) + if not found: + wandb.termerror(f"Could not find sweep {entity}/{project}/{resume_id}") + return + + if found.get("state") == "RUNNING": + wandb.termerror( + f"Cannot resume sweep {entity}/{project}/{resume_id}, it is already running" + ) + return + + sweep_obj_id = found["id"] + sweep_config = yaml.safe_load(found["config"]) + wandb.termlog(f"Resuming from existing sweep {entity}/{project}/{resume_id}") + if len(parsed_user_config.keys()) > 0: + wandb.termwarn( + "Sweep parameters loaded from resumed sweep, ignoring provided config" + ) + + prev_scheduler = json.loads(found.get("scheduler") or "{}") + run_spec = json.loads(prev_scheduler.get("run_spec", "{}")) + if ( + scheduler_job + and run_spec.get("job") + and run_spec.get("job") != scheduler_job + ): + wandb.termerror( + f"Resuming a launch sweep with a different scheduler job is not supported. Job loaded from sweep: {run_spec.get('job')}, job in config: {scheduler_job}" + ) + return + + prev_scheduler_args, prev_settings = sweep_utils.get_previous_args(run_spec) + # Passed in scheduler_args and settings override previous + scheduler_args.update(prev_scheduler_args) + settings.update(prev_settings) + if not queue: + wandb.termerror( + "Launch-sweeps require setting a 'queue', use --queue option or a 'queue' key in the 'launch' section in the config" + ) + return + + entrypoint = Scheduler.ENTRYPOINT if not scheduler_job else None + args = sweep_utils.construct_scheduler_args( + return_job=scheduler_job is not None, + sweep_config=sweep_config, + queue=queue, + project=project, + author=author, + ) + if not args: + return + + # validate training job existence + if not sweep_utils.check_job_exists(PublicApi(), sweep_config.get("job")): + return False + + # validate scheduler job existence, if present + if not sweep_utils.check_job_exists(PublicApi(), scheduler_job): + return False + + # Set run overrides for the Scheduler + overrides = {"run_config": {}} + if launch_args: + overrides["run_config"]["launch"] = launch_args + if scheduler_args: + overrides["run_config"]["scheduler"] = scheduler_args + if settings: + overrides["run_config"]["settings"] = settings + + if scheduler_job: + overrides["run_config"]["sweep_args"] = args + else: + overrides["args"] = args + + # configure scheduler job resource + resource = scheduler_args.get("resource") + if resource: + if resource == "local-process" and scheduler_job: + wandb.termerror( + "Scheduler jobs cannot be run with the 'local-process' resource" + ) + return + if resource == "local-process" and scheduler_args.get("docker_image"): + wandb.termerror( + "Scheduler jobs cannot be run with the 'local-process' resource and a docker image" + ) + return + else: # no resource set, default local-process if not scheduler job, else container + resource = "local-process" if not scheduler_job else "local-container" + + # Launch job spec for the Scheduler + launch_scheduler_spec = launch_utils.construct_launch_spec( + uri=Scheduler.PLACEHOLDER_URI, + api=api, + name="Scheduler.WANDB_SWEEP_ID", + project=project, + entity=entity, + docker_image=scheduler_args.get("docker_image"), + resource=resource, + entry_point=entrypoint, + resource_args=scheduler_args.get("resource_args", {}), + repository=launch_args.get("registry", {}).get("url", None), + job=scheduler_job, + version=None, + launch_config={"overrides": overrides}, + run_id="WANDB_SWEEP_ID", # scheduler inits run with sweep_id=run_id + author=None, # author gets passed into scheduler override args + ) + launch_scheduler_with_queue = json.dumps( + { + "queue": queue, + "run_queue_project": launch_utils.LAUNCH_DEFAULT_PROJECT, + "run_spec": json.dumps(launch_scheduler_spec), + } + ) + + sweep_id, warnings = api.upsert_sweep( + sweep_config, + project=project, + entity=entity, + obj_id=sweep_obj_id, # if resuming + launch_scheduler=launch_scheduler_with_queue, + state="PENDING", + prior_runs=prior_runs, + template_variable_values=scheduler_args.get("template_variables", None), + ) + sweep_utils.handle_sweep_config_violations(warnings) + # Log nicely formatted sweep information + styled_id = click.style(sweep_id, fg="yellow") + wandb.termlog(f"{'Resumed' if resume_id else 'Created'} sweep with ID: {styled_id}") + sweep_url = wandb_sdk.wandb_sweep._get_sweep_url(api, sweep_id) + if sweep_url: + styled_url = click.style(sweep_url, underline=True, fg="blue") + wandb.termlog(f"View sweep at: {styled_url}") + wandb.termlog(f"Scheduler added to launch queue ({queue})") + + +@cli.command(help=f"Launch or queue a W&B Job. See {url_registry.url('wandb-launch')}") +@click.option( + "--uri", + "-u", + metavar="(str)", + default=None, + help="Local path or git repo uri to launch. If provided this command will " + "create a job from the specified uri.", +) +@click.option( + "--job", + "-j", + metavar="(str)", + default=None, + help="Name of the job to launch. If passed in, launch does not require a uri.", +) +@click.option( + "--entry-point", + "-E", + metavar="NAME", + default=None, + help="""Entry point within project. [default: main]. If the entry point is not found, + attempts to run the project file with the specified name as a script, + using 'python' to run .py files and the default shell (specified by + environment variable $SHELL) to run .sh files. If passed in, will override the entrypoint value passed in using a config file.""", +) +@click.option( + "--git-version", + "-g", + metavar="GIT-VERSION", + hidden=True, + help="Version of the project to run, as a Git commit reference for Git projects.", +) +@click.option( + "--build-context", + metavar="(str)", + help="Path to the build context within the source code. Defaults to the " + "root of the source code. Compatible only with -u.", +) +@click.option( + "--job-name", + "-J", + metavar="(str)", + default=None, + hidden=True, + help="Name for the job created if the -u,--uri flag is passed in.", +) +@click.option( + "--name", + envvar="WANDB_NAME", + help="""Name of the run under which to launch the run. If not + specified, a random run name will be used to launch run. If passed in, will override the name passed in using a config file.""", +) +@click.option( + "--entity", + "-e", + metavar="(str)", + default=None, + help="""Name of the target entity which the new run will be sent to. Defaults to using the entity set by local wandb/settings folder. + If passed in, will override the entity value passed in using a config file.""", +) +@click.option( + "--project", + "-p", + metavar="(str)", + default=None, + help="""Name of the target project which the new run will be sent to. Defaults to using the project name given by the source uri + or for github runs, the git repo name. If passed in, will override the project value passed in using a config file.""", +) +@click.option( + "--resource", + "-r", + metavar="BACKEND", + default=None, + help="""Execution resource to use for run. Supported values: 'local-process', 'local-container', 'kubernetes', 'sagemaker', 'gcp-vertex'. + This is now a required parameter if pushing to a queue with no resource configuration. + If passed in, will override the resource value passed in using a config file.""", +) +@click.option( + "--docker-image", + "-d", + default=None, + metavar="DOCKER IMAGE", + help="""Specific docker image you'd like to use. In the form name:tag. + If passed in, will override the docker image value passed in using a config file.""", +) +@click.option( + "--base-image", + "-B", + default=None, + metavar="BASE IMAGE", + help="""Docker image to run job code in. Incompatible with --docker-image.""", +) +@click.option( + "--config", + "-c", + metavar="FILE", + help="""Path to JSON file (must end in '.json') or JSON string which will be passed + as a launch config. Dictation how the launched run will be configured.""", +) +@click.option( + "--set-var", + "-v", + "cli_template_vars", + default=None, + multiple=True, + help="""Set template variable values for queues with allow listing enabled, + as key-value pairs e.g. `--set-var key1=value1 --set-var key2=value2`""", +) +@click.option( + "--queue", + "-q", + is_flag=False, + flag_value="default", + default=None, + help="""Name of run queue to push to. If none, launches single run directly. If supplied without + an argument (`--queue`), defaults to queue 'default'. Else, if name supplied, specified run queue must exist under the + project and entity supplied.""", +) +@click.option( + "--async", + "run_async", + is_flag=True, + help="""Flag to run the job asynchronously. Defaults to false, i.e. unless --async is set, wandb launch will wait for + the job to finish. This option is incompatible with --queue; asynchronous options when running with an agent should be + set on wandb launch-agent.""", +) +@click.option( + "--resource-args", + "-R", + metavar="FILE", + help="""Path to JSON file (must end in '.json') or JSON string which will be passed + as resource args to the compute resource. The exact content which should be + provided is different for each execution backend. See documentation for layout of this file.""", +) +@click.option( + "--build", + "-b", + is_flag=True, + hidden=True, + help="Flag to build an associated job and push to queue as an image job.", +) +@click.option( + "--repository", + "-rg", + is_flag=False, + default=None, + hidden=True, + help="Name of a remote repository. Will be used to push a built image to.", +) +# TODO: this is only included for back compat. But we should remove this in the future +@click.option( + "--project-queue", + "-pq", + default=None, + hidden=True, + help="Name of the project containing the queue to push to. If none, defaults to entity level queues.", +) +@click.option( + "--dockerfile", + "-D", + default=None, + help="Path to the Dockerfile used to build the job, relative to the job's root", +) +@click.option( + "--priority", + "-P", + default=None, + type=click.Choice(["critical", "high", "medium", "low"]), + help="""When --queue is passed, set the priority of the job. Launch jobs with higher priority + are served first. The order, from highest to lowest priority, is: critical, high, medium, low""", +) +@display_error +def launch( + uri, + job, + entry_point, + git_version, + build_context, + name, + resource, + entity, + project, + docker_image, + base_image, + config, + cli_template_vars, + queue, + run_async, + resource_args, + build, + repository, + project_queue, + dockerfile, + priority, + job_name, +): + """Start a W&B run from the given URI. + + The URI can bea wandb URI, a GitHub repo uri, or a local path). In the case of a + wandb URI the arguments used in the original run will be used by default. These + arguments can be overridden using the args option, or specifying those arguments in + the config's 'overrides' key, 'args' field as a list of strings. + + Running `wandb launch [URI]` will launch the run directly. To add the run to a + queue, run `wandb launch [URI] --queue [optional queuename]`. + """ + logger.info( + f"=== Launch called with kwargs {locals()} CLI Version: {wandb.__version__}===" + ) + from wandb.sdk.launch._launch import _launch + from wandb.sdk.launch.create_job import _create_job + from wandb.sdk.launch.utils import _is_git_uri + + api = _get_cling_api() + wandb._sentry.configure_scope(process_context="launch_cli") + + if run_async and queue is not None: + raise LaunchError( + "Cannot use both --async and --queue with wandb launch, see help for details." + ) + + if queue and docker_image and not project: + raise LaunchError( + "Cannot use --queue and --docker together without a project. Please specify a project with --project or -p." + ) + + if priority is not None and queue is None: + raise LaunchError("--priority flag requires --queue to be set") + + if resource_args is not None: + resource_args = util.load_json_yaml_dict(resource_args) + if resource_args is None: + raise LaunchError("Invalid format for resource-args") + else: + resource_args = {} + + if entry_point is not None: + entry_point = shlex.split(entry_point) + + if config is not None: + config = util.load_json_yaml_dict(config) + if config is None: + raise LaunchError("Invalid format for config") + else: + config = {} + + resource = resource or config.get("resource") + + if build and queue is None: + raise LaunchError("Build flag requires a queue to be set") + + try: + launch_utils.check_logged_in(api) + except Exception: + wandb.termerror(f"Error running job: {traceback.format_exc()}") + + run_id = config.get("run_id") + + # If URI was provided, we need to create a job from it. + if uri: + if entry_point is None: + raise LaunchError( + "Cannot provide a uri without an entry point. Please provide an " + "entry point with --entry-point or -E." + ) + if job is not None: + raise LaunchError("Cannot provide both a uri and a job name.") + job_type = ( + "git" if _is_git_uri(uri) else "code" + ) # TODO: Add support for local URIs with git. + if entity is None: + entity = launch_utils.get_default_entity(api, config) + artifact, _, _ = _create_job( + api, + job_type, + uri, + entrypoint=" ".join(entry_point), + git_hash=git_version, + name=job_name, + project=project, + base_image=base_image, + build_context=build_context, + dockerfile=dockerfile, + entity=entity, + ) + if artifact is None: + raise LaunchError(f"Failed to create job from uri: {uri}") + job = f"{entity}/{project}/{artifact.name}" + + if dockerfile: + if "overrides" in config: + config["overrides"]["dockerfile"] = dockerfile + else: + config["overrides"] = {"dockerfile": dockerfile} + + if priority is not None: + priority_map = { + "critical": 0, + "high": 1, + "medium": 2, + "low": 3, + } + priority = priority_map[priority.lower()] + + template_variables = None + if cli_template_vars: + if queue is None: + raise LaunchError("'--set-var' flag requires queue to be set") + if entity is None: + entity = launch_utils.get_default_entity(api, config) + public_api = PublicApi() + runqueue = RunQueue(client=public_api.client, name=queue, entity=entity) + template_variables = launch_utils.fetch_and_validate_template_variables( + runqueue, cli_template_vars + ) + + if queue is None: + # direct launch + try: + run = asyncio.run( + _launch( + api, + job, + project=project, + entity=entity, + docker_image=docker_image, + name=name, + entry_point=entry_point, + version=git_version, + resource=resource, + resource_args=resource_args, + launch_config=config, + synchronous=(not run_async), + run_id=run_id, + repository=repository, + ) + ) + if asyncio.run(run.get_status()).state in [ + "failed", + "stopped", + "preempted", + ]: + wandb.termerror("Launched run exited with non-zero status") + sys.exit(1) + except LaunchError as e: + logger.exception("An error occurred.") + wandb._sentry.exception(e) + sys.exit(e) + except ExecutionError as e: + logger.exception("An error occurred.") + wandb._sentry.exception(e) + sys.exit(e) + except asyncio.CancelledError: + sys.exit(0) + else: + try: + _launch_add( + api, + job, + config, + template_variables, + project, + entity, + queue, + resource, + entry_point, + name, + git_version, + docker_image, + project_queue, + resource_args, + build=build, + run_id=run_id, + repository=repository, + priority=priority, + ) + + except Exception as e: + wandb._sentry.exception(e) + raise + + +@cli.command( + context_settings=CONTEXT, + help="Run a W&B launch agent.", +) +@click.pass_context +@click.option( + "--queue", + "-q", + "queues", + default=None, + multiple=True, + help="The name of a queue for the agent to watch. Multiple -q flags supported.", +) +@click.option( + "--entity", + "-e", + default=None, + help="The entity to use. Defaults to current logged-in user", +) +@click.option( + "--log-file", + "-l", + default=None, + help=( + "Destination for internal agent logs. Use - for stdout. " + "By default all agents logs will go to debug.log in your wandb/ " + "subdirectory or WANDB_DIR if set." + ), +) +@click.option( + "--max-jobs", + "-j", + default=None, + help="The maximum number of launch jobs this agent can run in parallel. Defaults to 1. Set to -1 for no upper limit", +) +@click.option( + "--config", "-c", default=None, help="path to the agent config yaml to use" +) +@click.option( + "--url", + "-u", + default=None, + hidden=True, + help="a wandb client registration URL, this is generated in the UI", +) +@click.option("--verbose", "-v", count=True, help="Display verbose output") +@display_error +def launch_agent( + ctx, + entity=None, + queues=None, + max_jobs=None, + config=None, + url=None, + log_file=None, + verbose=0, +): + logger.info( + f"=== Launch-agent called with kwargs {locals()} CLI Version: {wandb.__version__} ===" + ) + if url is not None: + raise LaunchError( + "--url is not supported in this version, upgrade with: pip install -u wandb" + ) + + import wandb.sdk.launch._launch as _launch + + if log_file is not None: + _launch.set_launch_logfile(log_file) + + api = _get_cling_api() + wandb._sentry.configure_scope(process_context="launch_agent") + agent_config, api = _launch.resolve_agent_config( + entity, max_jobs, queues, config, verbose + ) + + if len(agent_config.get("queues")) == 0: + raise LaunchError( + "To launch an agent please specify a queue or a list of queues in the configuration file or cli." + ) + + launch_utils.check_logged_in(api) + + wandb.termlog("Starting launch agent ✨") + try: + _launch.create_and_run_agent(api, agent_config) + except Exception as e: + wandb._sentry.exception(e) + raise + + +@cli.command(context_settings=CONTEXT, help="Run the W&B agent") +@click.pass_context +@click.option( + "--project", + "-p", + default=None, + help="""The name of the project where W&B runs created from the sweep are sent to. If the project is not specified, the run is sent to a project labeled 'Uncategorized'.""", +) +@click.option( + "--entity", + "-e", + default=None, + help="""The username or team name where you want to send W&B runs created by the sweep to. Ensure that the entity you specify already exists. If you don't specify an entity, the run will be sent to your default entity, which is usually your username.""", +) +@click.option( + "--count", default=None, type=int, help="The max number of runs for this agent." +) +@click.argument("sweep_id") +@display_error +def agent(ctx, project, entity, count, sweep_id): + api = _get_cling_api() + if not api.is_authenticated: + wandb.termlog("Login to W&B to use the sweep agent feature") + ctx.invoke(login, no_offline=True) + api = _get_cling_api(reset=True) + + wandb.termlog("Starting wandb agent 🕵️") + wandb_agent.agent(sweep_id, entity=entity, project=project, count=count) + + # you can send local commands like so: + # agent_api.command({'type': 'run', 'program': 'train.py', + # 'args': ['--max_epochs=10']}) + + +@cli.command( + context_settings=RUN_CONTEXT, help="Run a W&B launch sweep scheduler (Experimental)" +) +@click.pass_context +@click.argument("sweep_id") +@display_error +def scheduler( + ctx, + sweep_id, +): + api = InternalApi() + if not api.is_authenticated: + wandb.termlog("Login to W&B to use the sweep scheduler feature") + ctx.invoke(login, no_offline=True) + api = InternalApi(reset=True) + + wandb._sentry.configure_scope(process_context="sweep_scheduler") + wandb.termlog("Starting a Launch Scheduler 🚀") + from wandb.sdk.launch.sweeps import load_scheduler + + # TODO(gst): remove this monstrosity + # Future-proofing hack to pull any kwargs that get passed in through the CLI + kwargs = {} + for i, _arg in enumerate(ctx.args): + if isinstance(_arg, str) and _arg.startswith("--"): + # convert input kwargs from hyphens to underscores + _key = _arg[2:].replace("-", "_") + _args = ctx.args[i + 1] + if str.isdigit(_args): + _args = int(_args) + kwargs[_key] = _args + try: + sweep_type = kwargs.get("sweep_type", "wandb") + _scheduler = load_scheduler(scheduler_type=sweep_type)( + api, + sweep_id=sweep_id, + **kwargs, + ) + _scheduler.start() + except Exception as e: + wandb._sentry.exception(e) + raise + + +@cli.group(help="Commands for managing and viewing W&B jobs") +def job() -> None: + pass + + +@job.command("list", help="List jobs in a project") +@click.option( + "--project", + "-p", + envvar=env.PROJECT, + help="The project you want to list jobs from.", +) +@click.option( + "--entity", + "-e", + default="models", + envvar=env.ENTITY, + help="The entity the jobs belong to", +) +def _list(project, entity): + wandb.termlog(f"Listing jobs in {entity}/{project}") + public_api = PublicApi() + try: + jobs = public_api.list_jobs(entity=entity, project=project) + except wandb.errors.CommError as e: + wandb.termerror(f"{e}") + return + + if len(jobs) == 0: + wandb.termlog("No jobs found") + return + + for job in jobs: + aliases = [] + if len(job["edges"]) == 0: + # deleted? + continue + + name = job["edges"][0]["node"]["artifactSequence"]["name"] + for version in job["edges"]: + aliases += [x["alias"] for x in version["node"]["aliases"]] + + # only list the most recent 10 job versions + aliases_str = ",".join(aliases[::-1]) + wandb.termlog(f"{name} -- versions ({len(aliases)}): {aliases_str}") + + +@job.command( + help="Describe a launch job. Provide the launch job in the form of: entity/project/job-name:alias-or-version" +) +@click.argument("job") +def describe(job): + public_api = PublicApi() + try: + job = public_api.job(name=job) + except wandb.errors.CommError as e: + wandb.termerror(f"{e}") + return + + for key in job._job_info: + if key.startswith("_"): + continue + wandb.termlog(f"{key}: {job._job_info[key]}") + + +@job.command( + no_args_is_help=True, +) +@click.option( + "--project", + "-p", + envvar=env.PROJECT, + help="The project you want to list jobs from.", +) +@click.option( + "--entity", + "-e", + envvar=env.ENTITY, + help="The entity the jobs belong to", +) +@click.option( + "--name", + "-n", + help="Name for the job", +) +@click.option( + "--description", + "-d", + help="Description for the job", +) +@click.option( + "--alias", + "-a", + "aliases", + help="Alias for the job", + multiple=True, + default=tuple(), +) +@click.option( + "--entry-point", + "-E", + "entrypoint", + help="Entrypoint to the script, including an executable and an entrypoint " + "file. Required for code or repo jobs. If --build-context is provided, " + "paths in the entrypoint command will be relative to the build context.", +) +@click.option( + "--git-hash", + "-g", + "git_hash", + type=str, + help="Commit reference to use as the source for git jobs", +) +@click.option( + "--runtime", + "-r", + type=str, + help="Python runtime to execute the job", +) +@click.option( + "--build-context", + "-b", + type=str, + help="Path to the build context from the root of the job source code. If " + "provided, this is used as the base path for the Dockerfile and entrypoint.", +) +@click.option( + "--base-image", + "-B", + type=str, + help="Base image to use for the job. Incompatible with image jobs.", +) +@click.option( + "--dockerfile", + "-D", + type=str, + help="Path to the Dockerfile for the job. If --build-context is provided, " + "the Dockerfile path will be relative to the build context.", +) +@click.argument( + "job_type", + type=click.Choice(("git", "code", "image")), +) +@click.argument("path") +def create( + path, + project, + entity, + name, + job_type, + description, + aliases, + entrypoint, + git_hash, + runtime, + build_context, + base_image, + dockerfile, +): + """Create a job from a source, without a wandb run. + + Jobs can be of three types, git, code, or image. + + git: A git source, with an entrypoint either in the path or provided explicitly pointing to the main python executable. + code: A code path, containing a requirements.txt file. + image: A docker image. + """ + from wandb.sdk.launch.create_job import _create_job + + api = _get_cling_api() + wandb._sentry.configure_scope(process_context="job_create") + + entity = entity or os.getenv("WANDB_ENTITY") or api.default_entity + if not entity: + wandb.termerror("No entity provided, use --entity or set WANDB_ENTITY") + return + + project = project or os.getenv("WANDB_PROJECT") + if not project: + wandb.termerror("No project provided, use --project or set WANDB_PROJECT") + return + + if entrypoint is None and job_type in ["git", "code"]: + wandb.termwarn( + f"No entrypoint provided for {job_type} job, defaulting to main.py" + ) + entrypoint = "main.py" + + if job_type == "image" and base_image: + wandb.termerror("Cannot provide --base-image/-B for an `image` job") + return + + artifact, action, aliases = _create_job( + api=api, + path=path, + entity=entity, + project=project, + name=name, + job_type=job_type, + description=description, + aliases=list(aliases), + entrypoint=entrypoint, + git_hash=git_hash, + runtime=runtime, + build_context=build_context, + base_image=base_image, + dockerfile=dockerfile, + ) + if not artifact: + wandb.termerror("Job creation failed") + return + + artifact_path = f"{entity}/{project}/{artifact.name}" + msg = f"{action} job: {click.style(artifact_path, fg='yellow')}" + if len(aliases) == 1: + alias_str = click.style(aliases[0], fg="yellow") + msg += f", with alias: {alias_str}" + elif len(aliases) > 1: + alias_str = click.style(", ".join(aliases), fg="yellow") + msg += f", with aliases: {alias_str}" + + wandb.termlog(msg) + web_url = util.app_url(api.settings().get("base_url")) + url = click.style(f"{web_url}/{entity}/{project}/jobs", underline=True) + wandb.termlog(f"View all jobs in project '{project}' here: {url}\n") + + +@cli.command(context_settings=CONTEXT, help="Run the W&B local sweep controller") +@click.option("--verbose", is_flag=True, default=False, help="Display verbose output") +@click.argument("sweep_id") +@display_error +def controller(verbose, sweep_id): + click.echo("Starting wandb controller...") + from wandb import controller as wandb_controller + + tuner = wandb_controller(sweep_id) + tuner.run(verbose=verbose) + + +@cli.command(context_settings=RUN_CONTEXT, name="docker-run") +@click.pass_context +@click.argument("docker_run_args", nargs=-1) +def docker_run(ctx, docker_run_args): + """Wrap `docker run` and adds WANDB_API_KEY and WANDB_DOCKER environment variables. + + This will also set the runtime to nvidia if the nvidia-docker executable is present + on the system and --runtime wasn't set. + + See `docker run --help` for more details. + """ + import wandb.docker + + api = InternalApi() + args = list(docker_run_args) + if len(args) > 0 and args[0] == "run": + args.pop(0) + if len([a for a in args if a.startswith("--runtime")]) == 0 and _HAS_NVIDIA_DOCKER: + args = ["--runtime", "nvidia"] + args + # TODO: image_from_docker_args uses heuristics to find the docker image arg, there are likely cases + # where this won't work + image = util.image_from_docker_args(args) + resolved_image = None + if image: + resolved_image = wandb.docker.image_id(image) + if resolved_image: + args = ["-e", f"WANDB_DOCKER={resolved_image}"] + args + else: + wandb.termlog( + "Couldn't detect image argument, running command without the WANDB_DOCKER env variable" + ) + if api.api_key: + args = ["-e", f"WANDB_API_KEY={api.api_key}"] + args + else: + wandb.termlog( + "Not logged in, run `wandb login` from the host machine to enable result logging" + ) + subprocess.call(["docker", "run"] + args) + + +@cli.command(context_settings=RUN_CONTEXT) +@click.pass_context +@click.argument("docker_run_args", nargs=-1) +@click.argument("docker_image", required=False) +@click.option( + "--nvidia/--no-nvidia", + default=_HAS_NVIDIA_DOCKER, + help="Use the nvidia runtime, defaults to nvidia if nvidia-docker is present", +) +@click.option( + "--digest", is_flag=True, default=False, help="Output the image digest and exit" +) +@click.option( + "--jupyter/--no-jupyter", default=False, help="Run jupyter lab in the container" +) +@click.option( + "--dir", default="/app", help="Which directory to mount the code in the container" +) +@click.option("--no-dir", is_flag=True, help="Don't mount the current directory") +@click.option( + "--shell", default="/bin/bash", help="The shell to start the container with" +) +@click.option("--port", default="8888", help="The host port to bind jupyter on") +@click.option("--cmd", help="The command to run in the container") +@click.option( + "--no-tty", is_flag=True, default=False, help="Run the command without a tty" +) +@display_error +def docker( + ctx, + docker_run_args, + docker_image, + nvidia, + digest, + jupyter, + dir, + no_dir, + shell, + port, + cmd, + no_tty, +): + """Run your code in a docker container. + + W&B docker lets you run your code in a docker image ensuring wandb is configured. It + adds the WANDB_DOCKER and WANDB_API_KEY environment variables to your container and + mounts the current directory in /app by default. You can pass additional args which + will be added to `docker run` before the image name is declared, we'll choose a + default image for you if one isn't passed: + + ```sh + wandb docker -v /mnt/dataset:/app/data + wandb docker gcr.io/kubeflow-images-public/tensorflow-1.12.0-notebook-cpu:v0.4.0 --jupyter + wandb docker wandb/deepo:keras-gpu --no-tty --cmd "python train.py --epochs=5" + ``` + + By default, we override the entrypoint to check for the existence of wandb and + install it if not present. If you pass the --jupyter flag we will ensure jupyter is + installed and start jupyter lab on port 8888. If we detect nvidia-docker on your + system we will use the nvidia runtime. If you just want wandb to set environment + variable to an existing docker run command, see the wandb docker-run command. + """ + api = InternalApi() + if not _HAS_DOCKER: + raise ClickException("Docker not installed, install it from https://docker.com") + + import wandb.docker + + args = list(docker_run_args) + image = docker_image or "" + # remove run for users used to nvidia-docker + if len(args) > 0 and args[0] == "run": + args.pop(0) + if image == "" and len(args) > 0: + image = args.pop(0) + # If the user adds docker args without specifying an image (should be rare) + if not util.docker_image_regex(image.split("@")[0]): + if image: + args = args + [image] + image = wandb.docker.default_image(gpu=nvidia) + subprocess.call(["docker", "pull", image]) + _, repo_name, tag = wandb.docker.parse(image) + + resolved_image = wandb.docker.image_id(image) + if resolved_image is None: + raise ClickException( + f"Couldn't find image locally or in a registry, try running `docker pull {image}`" + ) + if digest: + sys.stdout.write(resolved_image) + exit(0) + + existing = wandb.docker.shell(["ps", "-f", f"ancestor={resolved_image}", "-q"]) + if existing: + if click.confirm( + "Found running container with the same image, do you want to attach?" + ): + subprocess.call(["docker", "attach", existing.split("\n")[0]]) + exit(0) + cwd = os.getcwd() + command = [ + "docker", + "run", + "-e", + "LANG=C.UTF-8", + "-e", + f"WANDB_DOCKER={resolved_image}", + "--ipc=host", + "-v", + wandb.docker.entrypoint + ":/wandb-entrypoint.sh", + "--entrypoint", + "/wandb-entrypoint.sh", + ] + if nvidia: + command.extend(["--runtime", "nvidia"]) + if not no_dir: + # TODO: We should default to the working directory if defined + command.extend(["-v", cwd + ":" + dir, "-w", dir]) + if api.api_key: + command.extend(["-e", f"WANDB_API_KEY={api.api_key}"]) + else: + wandb.termlog( + "Couldn't find WANDB_API_KEY, run `wandb login` to enable streaming metrics" + ) + if jupyter: + command.extend(["-e", "WANDB_ENSURE_JUPYTER=1", "-p", port + ":8888"]) + no_tty = True + cmd = f"jupyter lab --no-browser --ip=0.0.0.0 --allow-root --NotebookApp.token= --notebook-dir {dir}" + command.extend(args) + if no_tty: + command.extend([image, shell, "-c", cmd]) + else: + if cmd: + command.extend(["-e", f"WANDB_COMMAND={cmd}"]) + command.extend(["-it", image, shell]) + wandb.termlog("Launching docker container \U0001f6a2") + subprocess.call(command) + + +@cli.command( + context_settings=RUN_CONTEXT, + help="Start a local W&B container (deprecated, see wandb server --help)", + hidden=True, +) +@click.pass_context +@click.option("--port", "-p", default="8080", help="The host port to bind W&B local on") +@click.option( + "--env", "-e", default=[], multiple=True, help="Env vars to pass to wandb/local" +) +@click.option( + "--daemon/--no-daemon", default=True, help="Run or don't run in daemon mode" +) +@click.option( + "--upgrade", is_flag=True, default=False, help="Upgrade to the most recent version" +) +@click.option( + "--edge", is_flag=True, default=False, help="Run the bleeding edge", hidden=True +) +@display_error +def local(ctx, *args, **kwargs): + wandb.termwarn("`wandb local` has been replaced with `wandb server start`.") + ctx.invoke(start, *args, **kwargs) + + +@cli.group(help="Commands for operating a local W&B server") +def server(): + pass + + +@server.command(context_settings=RUN_CONTEXT, help="Start a local W&B server") +@click.pass_context +@click.option( + "--port", "-p", default="8080", help="The host port to bind W&B server on" +) +@click.option( + "--env", "-e", default=[], multiple=True, help="Env vars to pass to wandb/local" +) +@click.option( + "--daemon/--no-daemon", default=True, help="Run or don't run in daemon mode" +) +@click.option( + "--upgrade", + is_flag=True, + default=False, + help="Upgrade to the most recent version", + hidden=True, +) +@click.option( + "--edge", is_flag=True, default=False, help="Run the bleeding edge", hidden=True +) +@display_error +def start(ctx, port, env, daemon, upgrade, edge): + api = InternalApi() + if not _HAS_DOCKER: + raise ClickException("Docker not installed, install it from https://docker.com") + + import wandb.docker + + local_image_sha = wandb.docker.image_id("wandb/local").split("wandb/local")[-1] + registry_image_sha = wandb.docker.image_id_from_registry("wandb/local").split( + "wandb/local" + )[-1] + if local_image_sha != registry_image_sha: + if upgrade: + subprocess.call(["docker", "pull", "wandb/local"]) + else: + wandb.termlog( + "A new version of the W&B server is available, upgrade by calling `wandb server start --upgrade`" + ) + running = subprocess.check_output( + ["docker", "ps", "--filter", "name=^wandb-local$", "--format", "{{.ID}}"] + ) + if running != b"": + if upgrade: + subprocess.call(["docker", "stop", "wandb-local"]) + else: + wandb.termerror( + "A container named wandb-local is already running, run `docker stop wandb-local` if you want to start a new instance" + ) + exit(1) + image = "docker.pkg.github.com/wandb/core/local" if edge else "wandb/local" + username = getpass.getuser() + env_vars = ["-e", f"LOCAL_USERNAME={username}"] + for e in env: + env_vars.append("-e") + env_vars.append(e) + command = [ + "docker", + "run", + "--rm", + "-v", + "wandb:/vol", + "-p", + port + ":8080", + "--name", + "wandb-local", + ] + env_vars + host = f"http://localhost:{port}" + api.set_setting("base_url", host, globally=True, persist=True) + if daemon: + command += ["-d"] + command += [image] + + # DEVNULL is only in py3 + try: + from subprocess import DEVNULL + except ImportError: + DEVNULL = open(os.devnull, "wb") # noqa: N806 + code = subprocess.call(command, stdout=DEVNULL) + if daemon: + if code != 0: + wandb.termerror( + "Failed to launch the W&B server container, see the above error." + ) + exit(1) + else: + wandb.termlog(f"W&B server started at http://localhost:{port} \U0001f680") + wandb.termlog("You can stop the server by running `wandb server stop`") + if not api.api_key: + # Let the server start before potentially launching a browser + time.sleep(2) + ctx.invoke(login, host=host) + + +@server.command(context_settings=RUN_CONTEXT, help="Stop a local W&B server") +def stop(): + if not _HAS_DOCKER: + raise ClickException("Docker not installed, install it from https://docker.com") + subprocess.call(["docker", "stop", "wandb-local"]) + + +@cli.group(help="Commands for interacting with artifacts") +def artifact(): + pass + + +@artifact.command(context_settings=CONTEXT, help="Upload an artifact to wandb") +@click.argument("path") +@click.option( + "--name", "-n", help="The name of the artifact to push: project/artifact_name" +) +@click.option("--description", "-d", help="A description of this artifact") +@click.option("--type", "-t", default="dataset", help="The type of the artifact") +@click.option( + "--alias", + "-a", + default=["latest"], + multiple=True, + help="An alias to apply to this artifact", +) +@click.option("--id", "run_id", help="The run you want to upload to.") +@click.option( + "--resume", + is_flag=True, + default=None, + help="Resume the last run from your current directory.", +) +@click.option( + "--skip_cache", + is_flag=True, + default=False, + help="Skip caching while uploading artifact files.", +) +@click.option( + "--policy", + default="mutable", + type=click.Choice(["mutable", "immutable"]), + help="Set the storage policy while uploading artifact files.", +) +@display_error +def put( + path, + name, + description, + type, + alias, + run_id, + resume, + skip_cache, + policy, +): + if name is None: + name = os.path.basename(path) + public_api = PublicApi() + entity, project, artifact_name = public_api._parse_artifact_path(name) + if project is None: + project = click.prompt("Enter the name of the project you want to use") + # TODO: settings nightmare... + api = InternalApi() + api.set_setting("entity", entity) + api.set_setting("project", project) + artifact = wandb.Artifact(name=artifact_name, type=type, description=description) + artifact_path = f"{entity}/{project}/{artifact_name}:{alias[0]}" + if os.path.isdir(path): + wandb.termlog(f'Uploading directory {path} to: "{artifact_path}" ({type})') + artifact.add_dir(path, skip_cache=skip_cache, policy=policy) + elif os.path.isfile(path): + wandb.termlog(f'Uploading file {path} to: "{artifact_path}" ({type})') + artifact.add_file(path, skip_cache=skip_cache, policy=policy) + elif "://" in path: + wandb.termlog( + f'Logging reference artifact from {path} to: "{artifact_path}" ({type})' + ) + artifact.add_reference(path) + else: + raise ClickException("Path argument must be a file or directory") + + with wandb.init( + entity=entity, + project=project, + config={"path": path}, + job_type="cli_put", + id=run_id, + resume=resume, + ) as run: + run.log_artifact(artifact, aliases=alias) + artifact.wait() + + wandb.termlog( + "Artifact uploaded, use this artifact in a run by adding:\n", prefix=False + ) + wandb.termlog( + f' artifact = run.use_artifact("{artifact.source_qualified_name}")\n', + prefix=False, + ) + + +@artifact.command(context_settings=CONTEXT, help="Download an artifact from wandb") +@click.argument("path") +@click.option("--root", help="The directory you want to download the artifact to") +@click.option("--type", help="The type of artifact you are downloading") +@display_error +def get(path, root, type): + public_api = PublicApi() + entity, project, artifact_name = public_api._parse_artifact_path(path) + if project is None: + project = click.prompt("Enter the name of the project you want to use") + + try: + artifact_parts = artifact_name.split(":") + if len(artifact_parts) > 1: + version = artifact_parts[1] + artifact_name = artifact_parts[0] + else: + version = "latest" + if is_artifact_registry_project(project): + organization = path.split("/")[0] if path.count("/") == 2 else "" + # set entity to match the settings since in above code it was potentially set to an org + settings_entity = public_api.settings["entity"] or public_api.default_entity + # Registry artifacts are under the org entity. Because we offer a shorthand and alias for this path, + # we need to fetch the org entity to for the user behind the scenes. + entity = SDKInternalApi()._resolve_org_entity_name( + entity=settings_entity, organization=organization + ) + full_path = f"{entity}/{project}/{artifact_name}:{version}" + wandb.termlog( + "Downloading {type} artifact {full_path}".format( + type=type or "dataset", full_path=full_path + ) + ) + artifact = public_api.artifact(full_path, type=type) + path = artifact.download(root=root) + wandb.termlog(f"Artifact downloaded to {path}") + except ValueError: + raise ClickException("Unable to download artifact") + + +@artifact.command( + context_settings=CONTEXT, help="List all artifacts in a wandb project" +) +@click.argument("path") +@click.option("--type", "-t", help="The type of artifacts to list") +@display_error +def ls(path, type): + public_api = PublicApi() + if type is not None: + types = [public_api.artifact_type(type, path)] + else: + types = public_api.artifact_types(path) + + for kind in types: + for collection in kind.collections(): + versions = public_api.artifact_versions( + kind.type, + "/".join([kind.entity, kind.project, collection.name]), + per_page=1, + ) + latest = next(versions) + wandb.termlog( + f"{kind.type:<15s}{latest.updated_at:<15s}{util.to_human_size(latest.size):>15s} {latest.name:<20s}" + ) + + +@artifact.group(help="Commands for interacting with the artifact cache") +def cache(): + pass + + +@cache.command( + context_settings=CONTEXT, + help="Clean up less frequently used files from the artifacts cache", +) +@click.argument("target_size") +@click.option("--remove-temp/--no-remove-temp", default=False, help="Remove temp files") +@display_error +def cleanup(target_size, remove_temp): + target_size = util.from_human_size(target_size) + cache = get_artifact_file_cache() + reclaimed_bytes = cache.cleanup(target_size, remove_temp) + wandb.termlog(f"Reclaimed {util.to_human_size(reclaimed_bytes)} of space") + + +@cli.command(context_settings=CONTEXT, help="Pull files from Weights & Biases") +@click.argument("run", envvar=env.RUN_ID) +@click.option( + "--project", "-p", envvar=env.PROJECT, help="The project you want to download." +) +@click.option( + "--entity", + "-e", + default="models", + envvar=env.ENTITY, + help="The entity to scope the listing to.", +) +@display_error +def pull(run, project, entity): + api = InternalApi() + project, run = api.parse_slug(run, project=project) + urls = api.download_urls(project, run=run, entity=entity) + if len(urls) == 0: + raise ClickException("Run has no files") + click.echo(f"Downloading: {click.style(project, bold=True)}/{run}") + + for name in urls: + if api.file_current(name, urls[name]["md5"]): + click.echo(f"File {name} is up to date") + else: + length, response = api.download_file(urls[name]["url"]) + # TODO: I had to add this because some versions in CI broke click.progressbar + sys.stdout.write(f"File {name}\r") + dirname = os.path.dirname(name) + if dirname != "": + filesystem.mkdir_exists_ok(dirname) + with click.progressbar( + length=length, + label=f"File {name}", + fill_char=click.style("&", fg="green"), + ) as bar: + with open(name, "wb") as f: + for data in response.iter_content(chunk_size=4096): + f.write(data) + bar.update(len(data)) + + +@cli.command( + context_settings=CONTEXT, help="Restore code, config and docker state for a run" +) +@click.pass_context +@click.argument("run", envvar=env.RUN_ID) +@click.option("--no-git", is_flag=True, default=False, help="Don't restore git state") +@click.option( + "--branch/--no-branch", + default=True, + help="Whether to create a branch or checkout detached", +) +@click.option( + "--project", "-p", envvar=env.PROJECT, help="The project you wish to upload to." +) +@click.option( + "--entity", "-e", envvar=env.ENTITY, help="The entity to scope the listing to." +) +@display_error +def restore(ctx, run, no_git, branch, project, entity): + from wandb.old.core import wandb_dir + + api = _get_cling_api() + if ":" in run: + if "/" in run: + entity, rest = run.split("/", 1) + else: + rest = run + project, run = rest.split(":", 1) + elif run.count("/") > 1: + entity, run = run.split("/", 1) + + project, run = api.parse_slug(run, project=project) + commit, json_config, patch_content, metadata = api.run_config( + project, run=run, entity=entity + ) + repo = metadata.get("git", {}).get("repo") + image = metadata.get("docker") + restore_message = f"""`wandb restore` needs to be run from the same git repository as the original run. +Run `git clone {repo}` and restore from there or pass the --no-git flag.""" + if no_git: + commit = None + elif not api.git.enabled: + if repo: + raise ClickException(restore_message) + elif image: + wandb.termlog( + "Original run has no git history. Just restoring config and docker" + ) + + if commit and api.git.enabled: + wandb.termlog(f"Fetching origin and finding commit: {commit}") + subprocess.check_call(["git", "fetch", "--all"]) + try: + api.git.repo.commit(commit) + except ValueError: + wandb.termlog(f"Couldn't find original commit: {commit}") + commit = None + files = api.download_urls(project, run=run, entity=entity) + for filename in files: + if filename.startswith("upstream_diff_") and filename.endswith( + ".patch" + ): + commit = filename[len("upstream_diff_") : -len(".patch")] + try: + api.git.repo.commit(commit) + except ValueError: + commit = None + else: + break + + if commit: + wandb.termlog(f"Falling back to upstream commit: {commit}") + patch_path, _ = api.download_write_file(files[filename]) + else: + raise ClickException(restore_message) + else: + if patch_content: + patch_path = os.path.join(wandb_dir(), "diff.patch") + with open(patch_path, "w") as f: + f.write(patch_content) + else: + patch_path = None + + branch_name = f"wandb/{run}" + if branch and branch_name not in api.git.repo.branches: + api.git.repo.git.checkout(commit, b=branch_name) + wandb.termlog(f"Created branch {click.style(branch_name, bold=True)}") + elif branch: + wandb.termlog( + f"Using existing branch, run `git branch -D {branch_name}` from master for a clean checkout" + ) + api.git.repo.git.checkout(branch_name) + else: + wandb.termlog(f"Checking out {commit} in detached mode") + api.git.repo.git.checkout(commit) + + if patch_path: + # we apply the patch from the repository root so git doesn't exclude + # things outside the current directory + root = api.git.root + patch_rel_path = os.path.relpath(patch_path, start=root) + # --reject is necessary or else this fails any time a binary file + # occurs in the diff + exit_code = subprocess.call( + ["git", "apply", "--reject", patch_rel_path], cwd=root + ) + if exit_code == 0: + wandb.termlog("Applied patch") + else: + wandb.termerror( + "Failed to apply patch, try un-staging any un-committed changes" + ) + + filesystem.mkdir_exists_ok(wandb_dir()) + config_path = os.path.join(wandb_dir(), "config.yaml") + config = Config() + for k, v in json_config.items(): + if k not in ("_wandb", "wandb_version"): + config[k] = v + s = b"wandb_version: 1" + s += b"\n\n" + yaml.dump( + config._as_dict(), + Dumper=yaml.SafeDumper, + default_flow_style=False, + allow_unicode=True, + encoding="utf-8", + ) + s = s.decode("utf-8") + with open(config_path, "w") as f: + f.write(s) + + wandb.termlog(f"Restored config variables to {config_path}") + if image: + if not metadata["program"].startswith("<") and metadata.get("args") is not None: + # TODO: we may not want to default to python here. + runner = util.find_runner(metadata["program"]) or ["python"] + command = runner + [metadata["program"]] + metadata["args"] + cmd = " ".join(command) + else: + wandb.termlog("Couldn't find original command, just restoring environment") + cmd = None + wandb.termlog("Docker image found, attempting to start") + ctx.invoke(docker, docker_run_args=[image], cmd=cmd) + + return commit, json_config, patch_content, repo, metadata + + +@cli.command("online", help="Enable W&B sync") +@display_error +def online(): + api = InternalApi() + try: + api.clear_setting("mode", persist=True) + except configparser.Error: + pass + click.echo( + "W&B online. Running your script from this directory will now sync to the cloud." + ) + + +@cli.command("offline", help="Disable W&B sync") +@display_error +def offline(): + api = InternalApi() + try: + api.set_setting("mode", "offline", persist=True) + click.echo( + "W&B offline. Running your script from this directory will only write metadata locally. Use wandb disabled to completely turn off W&B." + ) + except configparser.Error: + click.echo( + "Unable to write config, copy and paste the following in your terminal to turn off W&B:\nexport WANDB_MODE=offline" + ) + + +@cli.command("on", hidden=True) +@click.pass_context +@display_error +def on(ctx): + ctx.invoke(online) + + +@cli.command("off", hidden=True) +@click.pass_context +@display_error +def off(ctx): + ctx.invoke(offline) + + +@cli.command("status", help="Show configuration settings") +@click.option( + "--settings/--no-settings", help="Show the current settings", default=True +) +def status(settings): + api = _get_cling_api() + if settings: + click.echo(click.style("Current Settings", bold=True)) + settings = api.settings() + click.echo( + json.dumps(settings, sort_keys=True, indent=2, separators=(",", ": ")) + ) + + +@cli.command("disabled", help="Disable W&B.") +@click.option( + "--service", + is_flag=True, + show_default=True, + default=True, + help="Disable W&B service", +) +def disabled(service): + api = InternalApi() + try: + api.set_setting("mode", "disabled", persist=True) + click.echo("W&B disabled.") + except configparser.Error: + click.echo( + "Unable to write config, copy and paste the following in your terminal to turn off W&B:\nexport WANDB_MODE=disabled" + ) + + +@cli.command("enabled", help="Enable W&B.") +@click.option( + "--service", + is_flag=True, + show_default=True, + default=True, + help="Enable W&B service", +) +def enabled(service): + api = InternalApi() + try: + api.set_setting("mode", "online", persist=True) + click.echo("W&B enabled.") + except configparser.Error: + click.echo( + "Unable to write config, copy and paste the following in your terminal to turn on W&B:\nexport WANDB_MODE=online" + ) + + +@cli.command(context_settings=CONTEXT, help="Verify your local instance") +@click.option("--host", default=None, help="Test a specific instance of W&B") +def verify(host): + # TODO: (kdg) Build this all into a WandbVerify object, and clean this up. + os.environ["WANDB_SILENT"] = "true" + os.environ["WANDB_PROJECT"] = "verify" + api = _get_cling_api() + reinit = False + if host is None: + host = api.settings("base_url") + wandb.termlog(f"Default host selected: {host}") + # if the given host does not match the default host, re-run init + elif host != api.settings("base_url"): + reinit = True + + tmp_dir = tempfile.mkdtemp() + wandb.termlog( + "Find detailed logs for this test at: {}".format(os.path.join(tmp_dir, "wandb")) + ) + os.chdir(tmp_dir) + os.environ["WANDB_BASE_URL"] = host + wandb.login(host=host) + if reinit: + api = _get_cling_api(reset=True) + if not wandb_verify.check_host(host): + sys.exit(1) + if not wandb_verify.check_logged_in(api, host): + sys.exit(1) + url_success, url = wandb_verify.check_graphql_put(api, host) + large_post_success = wandb_verify.check_large_post() + wandb_verify.check_secure_requests( + api.settings("base_url"), + "Checking requests to base url", + "Connections are not made over https. SSL required for secure communications.", + ) + if url: + wandb_verify.check_secure_requests( + url, + "Checking requests made over signed URLs", + "Signed URL requests not made over https. SSL is required for secure communications.", + ) + wandb_verify.check_cors_configuration(url, host) + wandb_verify.check_wandb_version(api) + check_run_success = wandb_verify.check_run(api) + check_artifacts_success = wandb_verify.check_artifacts() + check_sweeps_success = wandb_verify.check_sweeps(api) + if not ( + check_artifacts_success + and check_run_success + and large_post_success + and url_success + and check_sweeps_success + ): + sys.exit(1) + + +cli.add_command(beta) diff --git a/lib/python3.12/site-packages/wandb/errors/__init__.py b/lib/python3.12/site-packages/wandb/errors/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..1aa45ad8aced101f4bec07df6f1d8972dabffd34 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/errors/__init__.py @@ -0,0 +1,17 @@ +__all__ = ( + "Error", + "CommError", + "AuthenticationError", + "UsageError", + "UnsupportedError", + "WandbCoreNotAvailableError", +) + +from .errors import ( + AuthenticationError, + CommError, + Error, + UnsupportedError, + UsageError, + WandbCoreNotAvailableError, +) diff --git a/lib/python3.12/site-packages/wandb/errors/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/wandb/errors/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..9a36dcc75dc9066dae4443c86a7effd4b62db5c4 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/errors/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/errors/__pycache__/errors.cpython-312.pyc b/lib/python3.12/site-packages/wandb/errors/__pycache__/errors.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..0b0993d4aeb66e8c6850ec7271d3bf1d40fbff20 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/errors/__pycache__/errors.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/errors/__pycache__/links.cpython-312.pyc b/lib/python3.12/site-packages/wandb/errors/__pycache__/links.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..3f77bc45b37edfb0ebfe61a75883521974b2b111 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/errors/__pycache__/links.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/errors/__pycache__/term.cpython-312.pyc b/lib/python3.12/site-packages/wandb/errors/__pycache__/term.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..164e2540e8e0901ead9ee642aa381c5cad70feee Binary files /dev/null and b/lib/python3.12/site-packages/wandb/errors/__pycache__/term.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/errors/__pycache__/util.cpython-312.pyc b/lib/python3.12/site-packages/wandb/errors/__pycache__/util.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e598030a383d1aee0e3abc124aba55e8eca0861c Binary files /dev/null and b/lib/python3.12/site-packages/wandb/errors/__pycache__/util.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/errors/__pycache__/warnings.cpython-312.pyc b/lib/python3.12/site-packages/wandb/errors/__pycache__/warnings.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..3692f83a9c93ca692cc01057175e57e7876d5c68 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/errors/__pycache__/warnings.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/errors/errors.py b/lib/python3.12/site-packages/wandb/errors/errors.py new file mode 100644 index 0000000000000000000000000000000000000000..e612dc0f3551fd911582e8a9cf8eeb1535f8d5bc --- /dev/null +++ b/lib/python3.12/site-packages/wandb/errors/errors.py @@ -0,0 +1,37 @@ +from typing import Optional + + +class Error(Exception): + """Base W&B Error.""" + + def __init__(self, message, context: Optional[dict] = None) -> None: + super().__init__(message) + self.message = message + # sentry context capture + if context: + self.context = context + + +class CommError(Error): + """Error communicating with W&B servers.""" + + def __init__(self, msg, exc=None) -> None: + self.exc = exc + self.message = msg + super().__init__(self.message) + + +class AuthenticationError(CommError): + """Raised when authentication fails.""" + + +class UsageError(Error): + """Raised when an invalid usage of the SDK API is detected.""" + + +class UnsupportedError(UsageError): + """Raised when trying to use a feature that is not supported.""" + + +class WandbCoreNotAvailableError(Error): + """Raised when wandb core is not available.""" diff --git a/lib/python3.12/site-packages/wandb/errors/links.py b/lib/python3.12/site-packages/wandb/errors/links.py new file mode 100644 index 0000000000000000000000000000000000000000..7ff87cde68e779ab1d857a347dfbf6bb6445e74e --- /dev/null +++ b/lib/python3.12/site-packages/wandb/errors/links.py @@ -0,0 +1,73 @@ +"""Module containing the WBURLs class and WBURL dataclass. + +Used to store predefined URLs that can be associated with a name. The URLs are +shortened using with the `wandb.me` domain, using dub.co as the shortening service. +If the URLs need to be updates, use the dub.co service to point to the new URL. +""" + +from __future__ import annotations + +from dataclasses import dataclass + + +@dataclass +class WBURL: + url: str + description: str + + +class Registry: + """A collection of URLs that can be associated with a name.""" + + def __init__(self) -> None: + self.urls: dict[str, WBURL] = { + "wandb-launch": WBURL( + "https://wandb.me/launch", + "Link to the W&B launch marketing page", + ), + "wandb-init": WBURL( + "https://wandb.me/wandb-init", + "Link to the wandb.init reference documentation page", + ), + "define-metric": WBURL( + "https://wandb.me/define-metric", + "Link to the W&B developer guide documentation page on wandb.define_metric", + ), + "developer-guide": WBURL( + "https://wandb.me/developer-guide", + "Link to the W&B developer guide top level page", + ), + "wandb-core": WBURL( + "https://wandb.me/wandb-core", + "Link to the documentation for the wandb-core service", + ), + "wandb-server": WBURL( + "https://wandb.me/wandb-server", + "Link to the documentation for the self-hosted W&B server", + ), + "multiprocess": WBURL( + "https://wandb.me/multiprocess", + ( + "Link to the W&B developer guide documentation page on how to " + "use wandb in a multiprocess environment" + ), + ), + } + + def url(self, name: str) -> str: + """Get the URL associated with the given name.""" + wb_url = self.urls.get(name) + if wb_url: + return wb_url.url + raise ValueError(f"URL not found for {name}") + + def description(self, name: str) -> str: + """Get the description associated with the given name.""" + wb_url = self.urls.get(name) + if wb_url: + return wb_url.description + raise ValueError(f"Description not found for {name}") + + +# This is an instance of the Links class that can be used to access the URLs +url_registry = Registry() diff --git a/lib/python3.12/site-packages/wandb/errors/term.py b/lib/python3.12/site-packages/wandb/errors/term.py new file mode 100644 index 0000000000000000000000000000000000000000..78ddeced609bb33cfd9486d0b0ac409865c9a243 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/errors/term.py @@ -0,0 +1,415 @@ +"""Global functions for printing to stderr for wandb.""" + +from __future__ import annotations + +import contextlib +import logging +import os +import re +import shutil +import sys +import threading +from typing import TYPE_CHECKING, Iterator, Protocol + +import click + +if TYPE_CHECKING: + import wandb + +LOG_STRING = click.style("wandb", fg="blue", bold=True) +LOG_STRING_NOCOLOR = "wandb" +ERROR_STRING = click.style("ERROR", bg="red", fg="green") +WARN_STRING = click.style("WARNING", fg="yellow") + +_silent: bool = False +"""If true, _logger is used instead of printing to stderr.""" + +_logger: SupportsLeveledLogging | None = None +"""A fallback logger for _silent mode.""" + +_show_info: bool = True +"""If false, then termlog() uses silent mode (see _silent).""" + +_show_warnings: bool = True +"""If false, then termwarn() uses silent mode (see _silent).""" + +_show_errors: bool = True +"""If false, then termerror() uses silent mode (see _silent).""" + + +_printed_messages: set[str] = set() +"""Messages logged with repeat=False.""" + +_dynamic_text_lock = threading.Lock() +"""Lock held for dynamic text operations. + +All uses of `_dynamic_blocks` and calls to functions that start with +the `_l_` prefix must be guarded by this lock. +""" + +_dynamic_blocks: list[DynamicBlock] = [] +"""Active dynamic text areas, created with dynamic_text().""" + + +class SupportsLeveledLogging(Protocol): + """Portion of the standard logging.Logger used in this module.""" + + def info(self, msg: str) -> None: ... + def warning(self, msg: str) -> None: ... + def error(self, msg: str) -> None: ... + + +def termsetup( + settings: wandb.Settings, + logger: SupportsLeveledLogging | None, +) -> None: + """Configure the global logging functions. + + Args: + settings: The settings object passed to wandb.setup() or wandb.init(). + logger: A fallback logger to use for "silent" mode. In this mode, + the logger is used instead of printing to stderr. + """ + global _silent, _show_info, _show_warnings, _show_errors, _logger + _silent = settings.silent + _show_info = settings.show_info + _show_warnings = settings.show_warnings + _show_errors = settings.show_errors + _logger = logger + + +@contextlib.contextmanager +def dynamic_text() -> Iterator[DynamicBlock | None]: + """A context manager that provides a handle to a new dynamic text area. + + The text goes to stderr. Returns None if dynamic text is not supported. + + Dynamic text must only be used while `wandb` has control of the terminal, + or else text written by other programs will be overwritten. It's + appropriate to use during a blocking operation. + + ``` + with term.dynamic_text() as text_area: + if text_area: + text_area.set_text("Writing to a terminal.") + for i in range(2000): + text_area.set_text(f"Still going... ({i}/2000)") + time.sleep(0.001) + else: + wandb.termlog("Writing to a file or dumb terminal.") + time.sleep(1) + wandb.termlog("Finished 1000/2000 tasks, still working...") + time.sleep(1) + wandb.termlog("Done!", err=True) + ``` + """ + # For now, dynamic text always corresponds to the "INFO" level. + if _silent or not _show_info: + yield None + return + + # NOTE: In Jupyter notebooks, this will return False. Notebooks + # support ANSI color sequences and the '\r' character, but not + # cursor motions or line clear commands. + if not _sys_stderr_isatty(): + yield None + return + + # This is a convention to indicate that the terminal doesn't support + # clearing the screen / positioning the cursor. + if os.environ.get("TERM") == "dumb": + yield None + return + + # NOTE: On Windows < 10, ANSI escape sequences such as \x1b[Am and \x1b[2K, + # used to move the cursor and clear text, aren't supported by the built-in + # console. However, we rely on the click library's use of colorama which + # emulates support for such sequences. + # + # For this reason, we don't have special checks for Windows. + + block = DynamicBlock() + + with _dynamic_text_lock: + _dynamic_blocks.append(block) + + try: + yield block + finally: + with _dynamic_text_lock: + block._lines_to_print = [] + _l_rerender_dynamic_blocks() + _dynamic_blocks.remove(block) + + +def _sys_stderr_isatty() -> bool: + """Returns sys.stderr.isatty(). + + Defined here for patching in tests. + """ + return sys.stderr.isatty() + + +def termlog( + string: str = "", + newline: bool = True, + repeat: bool = True, + prefix: bool = True, +) -> None: + r"""Log an informational message to stderr. + + The message may contain ANSI color sequences and the \n character. + Colors are stripped if stderr is not a TTY. + + Args: + string: The message to display. + newline: Whether to add a newline to the end of the string. + repeat: If false, then the string is not printed if an exact match has + already been printed through any of the other logging functions + in this file. + prefix: Whether to include the 'wandb:' prefix. + """ + _log( + string, + newline=newline, + repeat=repeat, + prefix=prefix, + silent=not _show_info, + ) + + +def termwarn( + string: str, + newline: bool = True, + repeat: bool = True, + prefix: bool = True, +) -> None: + """Log a warning to stderr. + + The arguments are the same as for `termlog()`. + """ + string = "\n".join([f"{WARN_STRING} {s}" for s in string.split("\n")]) + _log( + string, + newline=newline, + repeat=repeat, + prefix=prefix, + silent=not _show_warnings, + level=logging.WARNING, + ) + + +def termerror( + string: str, + newline: bool = True, + repeat: bool = True, + prefix: bool = True, +) -> None: + """Log an error to stderr. + + The arguments are the same as for `termlog()`. + """ + string = "\n".join([f"{ERROR_STRING} {s}" for s in string.split("\n")]) + _log( + string, + newline=newline, + repeat=repeat, + prefix=prefix, + silent=not _show_errors, + level=logging.ERROR, + ) + + +class DynamicBlock: + """A handle to a changeable text area in the terminal.""" + + def __init__(self): + self._lines_to_print = [] + self._num_lines_printed = 0 + + def set_text(self, text: str, prefix=True) -> None: + r"""Replace the text in this block. + + Args: + text: The text to put in the block, with lines separated + by \n characters. The text should not end in \n unless + a blank line at the end of the block is desired. + prefix: Whether to include the "wandb:" prefix. + """ + with _dynamic_text_lock: + self._lines_to_print = text.splitlines() + + if prefix: + self._lines_to_print = [ + f"{LOG_STRING}: {line}" for line in self._lines_to_print + ] + + _l_rerender_dynamic_blocks() + + def _l_clear(self) -> None: + """Send terminal commands to clear all previously printed lines. + + The lock must be held, and the cursor must be on the line after this + block of text. + """ + # NOTE: We rely on the fact that click.echo() uses colorama which + # emulates these ANSI sequences on older Windows versions. + # + # \r move cursor to start of line + # \x1b[Am move cursor up + # \x1b[2K delete line (sometimes moves cursor) + # \r move cursor to start of line + move_up_and_delete_line = "\r\x1b[Am\x1b[2K\r" + click.echo( + move_up_and_delete_line * self._num_lines_printed, + file=sys.stderr, + nl=False, + ) + self._num_lines_printed = 0 + + def _l_print(self) -> None: + """Print out this block of text. + + The lock must be held. + """ + if self._lines_to_print: + # Trim lines before printing. This is crucial because the \x1b[Am + # (cursor up) sequence used when clearing the text moves up by one + # visual line, and the terminal may be wrapping long lines onto + # multiple visual lines. + # + # There is no ANSI escape sequence that moves the cursor up by one + # "physical" line instead. Note that the user may resize their + # terminal. + term_width = _shutil_get_terminal_width() + click.echo( + "\n".join( + _ansi_shorten(line, term_width) # + for line in self._lines_to_print + ), + file=sys.stderr, + ) + + self._num_lines_printed += len(self._lines_to_print) + + +def _shutil_get_terminal_width() -> int: + """Returns the width of the terminal. + + Defined here for patching in tests. + """ + columns, _ = shutil.get_terminal_size() + return columns + + +_ANSI_RE = re.compile("\x1b\\[(K|.*?m)") + + +def _ansi_shorten(text: str, width: int) -> str: + """Shorten text potentially containing ANSI sequences to fit a width.""" + first_ansi = _ANSI_RE.search(text) + + if not first_ansi: + return _raw_shorten(text, width) + + if first_ansi.start() > width - 3: + return _raw_shorten(text[: first_ansi.start()], width) + + return text[: first_ansi.end()] + _ansi_shorten( + text[first_ansi.end() :], + # Key part: the ANSI sequence doesn't reduce the remaining width. + width - first_ansi.start(), + ) + + +def _raw_shorten(text: str, width: int) -> str: + """Shorten text to fit a width, replacing the end with "...". + + Unlike textwrap.shorten(), this does not drop whitespace or do anything + smart. + """ + if len(text) <= width: + return text + + return text[: width - 3] + "..." + + +def _log( + string="", + newline=True, + repeat=True, + prefix=True, + silent=False, + level=logging.INFO, +) -> None: + with _dynamic_text_lock, _l_above_dynamic_text(): + if not repeat: + if string in _printed_messages: + return + + if len(_printed_messages) < 1000: + _printed_messages.add(string) + + if prefix: + string = "\n".join([f"{LOG_STRING}: {s}" for s in string.split("\n")]) + + silent = silent or _silent + if not silent: + click.echo(string, file=sys.stderr, nl=newline) + elif not _logger: + pass # No fallback logger, so nothing to do. + elif level == logging.ERROR: + _logger.error(click.unstyle(string)) + elif level == logging.WARNING: + _logger.warning(click.unstyle(string)) + else: + _logger.info(click.unstyle(string)) + + +def _l_rerender_dynamic_blocks() -> None: + """Clear and re-print all dynamic text. + + The lock must be held. The cursor must be positioned at the start of + the first line after the dynamic text area. + """ + with _l_above_dynamic_text(): + # We just want the side-effect of rerendering the dynamic text. + pass + + +@contextlib.contextmanager +def _l_above_dynamic_text(): + """A context manager for inserting static text above any dynamic text. + + The lock must be held. The cursor must be positioned at the start of the + first line after the dynamic text area. + + The dynamic text is re-rendered. + """ + _l_clear_dynamic_blocks() + + try: + yield + finally: + _l_print_dynamic_blocks() + + +def _l_clear_dynamic_blocks() -> None: + """Delete all dynamic text. + + The lock must be held, and the cursor must be positioned at the start + of the first line after the dynamic text area. After this, the cursor + is positioned at the start of the first line after all static text. + """ + for block in reversed(_dynamic_blocks): + block._l_clear() + + +def _l_print_dynamic_blocks() -> None: + """Output all dynamic text. + + The lock must be held. After this, the cursor is positioned at the start + of the first line after the dynamic text area. + """ + for block in _dynamic_blocks: + block._l_print() diff --git a/lib/python3.12/site-packages/wandb/errors/util.py b/lib/python3.12/site-packages/wandb/errors/util.py new file mode 100644 index 0000000000000000000000000000000000000000..0dd207c9e59f6ad15405b753c29f8b241238941a --- /dev/null +++ b/lib/python3.12/site-packages/wandb/errors/util.py @@ -0,0 +1,57 @@ +from typing import Optional + +from wandb.proto import wandb_internal_pb2 as pb + +from . import AuthenticationError, CommError, Error, UnsupportedError, UsageError + +to_exception_map = { + pb.ErrorInfo.UNKNOWN: Error, + pb.ErrorInfo.COMMUNICATION: CommError, + pb.ErrorInfo.AUTHENTICATION: AuthenticationError, + pb.ErrorInfo.USAGE: UsageError, + pb.ErrorInfo.UNSUPPORTED: UnsupportedError, +} + +from_exception_map = {v: k for k, v in to_exception_map.items()} + + +class ProtobufErrorHandler: + """Converts protobuf errors to exceptions and vice versa.""" + + @staticmethod + def to_exception(error: pb.ErrorInfo) -> Optional[Error]: + """Convert a protobuf error to an exception. + + Args: + error: The protobuf error to convert. + + Returns: + The corresponding exception. + + """ + if not error.SerializeToString(): + return None + + if error.code in to_exception_map: + return to_exception_map[error.code](error.message) + return Error(error.message) + + @classmethod + def from_exception(cls, exc: Error) -> "pb.ErrorInfo": + """Convert an wandb error to a protobuf error message. + + Args: + exc: The exception to convert. + + Returns: + The corresponding protobuf error message. + """ + if not isinstance(exc, Error): + raise TypeError("exc must be a subclass of wandb.errors.Error") + + code = None + for subclass in type(exc).__mro__: + if subclass in from_exception_map: + code = from_exception_map[subclass] # type: ignore + break + return pb.ErrorInfo(code=code, message=str(exc)) # type: ignore diff --git a/lib/python3.12/site-packages/wandb/errors/warnings.py b/lib/python3.12/site-packages/wandb/errors/warnings.py new file mode 100644 index 0000000000000000000000000000000000000000..f956757b378de6f5481aeb467961013ad264f3a9 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/errors/warnings.py @@ -0,0 +1,2 @@ +class WandbWarning(Warning): + """Base W&B Warning.""" diff --git a/lib/python3.12/site-packages/wandb/vendor/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..5b8db8bd57648cb07975f0c44b2085cc0b99c9f6 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..64558d4f36f01ce78b6cc0e9e5bdf46e8bfe4796 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/events.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/events.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e62f3271e6133c171a15d6ba197f2f5a1e382209 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/events.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/patterns.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/patterns.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f1fec1c2c48fa30121b874e9b8de487964ae16ff Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/patterns.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/version.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/version.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..162b31efcc314debb1194475cfc250b6ada91c6d Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/version.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/watchmedo.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/watchmedo.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..df51172223f41212c194d40fda660619ed63da41 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/__pycache__/watchmedo.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c40ba2b4bbd0116e366dbb141f291a8ba2dbbdfe Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/api.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/api.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..98e2fdb09c44c6394ec244c9b8150c6e280939f8 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/api.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/fsevents.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/fsevents.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..59f639b98c8eacbe5c8610c98adb9639eb9779dd Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/fsevents.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/fsevents2.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/fsevents2.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a6f4153f87d66d5c21a8fde10e070c09a4bad36b Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/fsevents2.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/inotify_buffer.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/inotify_buffer.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..568af5f0df66c6d957c8a0951150ee923ef0246a Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/inotify_buffer.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/inotify_c.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/inotify_c.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7461db1abacf9968a0412658fbd2b54e26ad30a4 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/inotify_c.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/kqueue.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/kqueue.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..41212fde3d9adec55387ce0cb8d1c06811099e2a Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/kqueue.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/polling.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/polling.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e444dee7ff0df20f96be7e42805aaa4fdd04c874 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/polling.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/read_directory_changes.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/read_directory_changes.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..dc08caa21978fff3d8e9a7dc4e5e57a858be695d Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/observers/__pycache__/read_directory_changes.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/tricks/__init__.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/tricks/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..cdcc14b226ff8364f60fa6b2e3eae1f3834587db --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/tricks/__init__.py @@ -0,0 +1,174 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# +# Copyright 2011 Yesudeep Mangalapilly +# Copyright 2012 Google, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + + +import os +import signal +import subprocess +import time + +from wandb_watchdog.utils import echo, has_attribute +from wandb_watchdog.events import PatternMatchingEventHandler + + +class Trick(PatternMatchingEventHandler): + + """Your tricks should subclass this class.""" + + @classmethod + def generate_yaml(cls): + context = dict(module_name=cls.__module__, + klass_name=cls.__name__) + template_yaml = """- %(module_name)s.%(klass_name)s: + args: + - argument1 + - argument2 + kwargs: + patterns: + - "*.py" + - "*.js" + ignore_patterns: + - "version.py" + ignore_directories: false +""" + return template_yaml % context + + +class LoggerTrick(Trick): + + """A simple trick that does only logs events.""" + + def on_any_event(self, event): + pass + + @echo.echo + def on_modified(self, event): + pass + + @echo.echo + def on_deleted(self, event): + pass + + @echo.echo + def on_created(self, event): + pass + + @echo.echo + def on_moved(self, event): + pass + + +class ShellCommandTrick(Trick): + + """Executes shell commands in response to matched events.""" + + def __init__(self, shell_command=None, patterns=None, ignore_patterns=None, + ignore_directories=False, wait_for_process=False, + drop_during_process=False): + super(ShellCommandTrick, self).__init__(patterns, ignore_patterns, + ignore_directories) + self.shell_command = shell_command + self.wait_for_process = wait_for_process + self.drop_during_process = drop_during_process + self.process = None + + def on_any_event(self, event): + from string import Template + + if self.drop_during_process and self.process and self.process.poll() is None: + return + + if event.is_directory: + object_type = 'directory' + else: + object_type = 'file' + + context = { + 'watch_src_path': event.src_path, + 'watch_dest_path': '', + 'watch_event_type': event.event_type, + 'watch_object': object_type, + } + + if self.shell_command is None: + if has_attribute(event, 'dest_path'): + context.update({'dest_path': event.dest_path}) + command = 'echo "${watch_event_type} ${watch_object} from ${watch_src_path} to ${watch_dest_path}"' + else: + command = 'echo "${watch_event_type} ${watch_object} ${watch_src_path}"' + else: + if has_attribute(event, 'dest_path'): + context.update({'watch_dest_path': event.dest_path}) + command = self.shell_command + + command = Template(command).safe_substitute(**context) + self.process = subprocess.Popen(command, shell=True) + if self.wait_for_process: + self.process.wait() + + +class AutoRestartTrick(Trick): + + """Starts a long-running subprocess and restarts it on matched events. + + The command parameter is a list of command arguments, such as + ['bin/myserver', '-c', 'etc/myconfig.ini']. + + Call start() after creating the Trick. Call stop() when stopping + the process. + """ + + def __init__(self, command, patterns=None, ignore_patterns=None, + ignore_directories=False, stop_signal=signal.SIGINT, + kill_after=10): + super(AutoRestartTrick, self).__init__( + patterns, ignore_patterns, ignore_directories) + self.command = command + self.stop_signal = stop_signal + self.kill_after = kill_after + self.process = None + + def start(self): + self.process = subprocess.Popen(self.command, preexec_fn=os.setsid) + + def stop(self): + if self.process is None: + return + try: + os.killpg(os.getpgid(self.process.pid), self.stop_signal) + except OSError: + # Process is already gone + pass + else: + kill_time = time.time() + self.kill_after + while time.time() < kill_time: + if self.process.poll() is not None: + break + time.sleep(0.25) + else: + try: + os.killpg(os.getpgid(self.process.pid), 9) + except OSError: + # Process is already gone + pass + self.process = None + + @echo.echo + def on_any_event(self, event): + self.stop() + self.start() diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/tricks/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/tricks/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..98f7a6442540fc673990b38b3794e09c4a6ad398 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/tricks/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__init__.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__init__.py new file mode 100644 index 0000000000000000000000000000000000000000..0e1b25803132190a2338f1f95ae6910295d2e473 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__init__.py @@ -0,0 +1,151 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# +# Copyright 2011 Yesudeep Mangalapilly +# Copyright 2012 Google, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + + +""" +:module: watchdog.utils +:synopsis: Utility classes and functions. +:author: yesudeep@google.com (Yesudeep Mangalapilly) + +Classes +------- +.. autoclass:: BaseThread + :members: + :show-inheritance: + :inherited-members: + +""" +import os +import sys +import threading +from wandb_watchdog.utils import platform +from wandb_watchdog.utils.compat import Event + + +if sys.version_info[0] == 2 and platform.is_windows(): + # st_ino is not implemented in os.stat on this platform + import win32stat + stat = win32stat.stat +else: + stat = os.stat + + +def has_attribute(ob, attribute): + """ + :func:`hasattr` swallows exceptions. :func:`has_attribute` tests a Python object for the + presence of an attribute. + + :param ob: + object to inspect + :param attribute: + ``str`` for the name of the attribute. + """ + return getattr(ob, attribute, None) is not None + + +class UnsupportedLibc(Exception): + pass + + +class BaseThread(threading.Thread): + """ Convenience class for creating stoppable threads. """ + + def __init__(self): + threading.Thread.__init__(self) + self.daemon = True + self._stopped_event = Event() + + @property + def stopped_event(self): + return self._stopped_event + + def should_keep_running(self): + """Determines whether the thread should continue running.""" + return not self._stopped_event.is_set() + + def on_thread_stop(self): + """Override this method instead of :meth:`stop()`. + :meth:`stop()` calls this method. + + This method is called immediately after the thread is signaled to stop. + """ + pass + + def stop(self): + """Signals the thread to stop.""" + self._stopped_event.set() + self.on_thread_stop() + + def on_thread_start(self): + """Override this method instead of :meth:`start()`. :meth:`start()` + calls this method. + + This method is called right before this thread is started and this + object’s run() method is invoked. + """ + pass + + def start(self): + self.on_thread_start() + threading.Thread.start(self) + + +def load_module(module_name): + """Imports a module given its name and returns a handle to it.""" + try: + __import__(module_name) + except ImportError: + raise ImportError('No module named %s' % module_name) + return sys.modules[module_name] + + +def load_class(dotted_path): + """Loads and returns a class definition provided a dotted path + specification the last part of the dotted path is the class name + and there is at least one module name preceding the class name. + + Notes: + You will need to ensure that the module you are trying to load + exists in the Python path. + + Examples: + - module.name.ClassName # Provided module.name is in the Python path. + - module.ClassName # Provided module is in the Python path. + + What won't work: + - ClassName + - modle.name.ClassName # Typo in module name. + - module.name.ClasNam # Typo in classname. + """ + dotted_path_split = dotted_path.split('.') + if len(dotted_path_split) > 1: + klass_name = dotted_path_split[-1] + module_name = '.'.join(dotted_path_split[:-1]) + + module = load_module(module_name) + if has_attribute(module, klass_name): + klass = getattr(module, klass_name) + return klass + # Finally create and return an instance of the class + # return klass(*args, **kwargs) + else: + raise AttributeError('Module %s does not have class attribute %s' % ( + module_name, klass_name)) + else: + raise ValueError( + 'Dotted module path %s must contain a module name and a classname' % dotted_path) diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/__init__.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/__init__.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..059195405e25aea436a0ab7c412bf72434fc51e1 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/__init__.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/bricks.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/bricks.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e3d59f009ed63e6196cbfc512122bbffc9269bda Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/bricks.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/compat.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/compat.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..1e28c2a58a515d9acf6011aa3ea90d1038c06136 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/compat.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/decorators.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/decorators.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ab8e579f1a54c1bf8eac23b8311f1608fd10087d Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/decorators.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/delayed_queue.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/delayed_queue.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f8f58065d8eec5427a9ef148ad01438cc6f5f368 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/delayed_queue.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/dirsnapshot.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/dirsnapshot.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..166b2498d327b0e3b62318fabfebdb1c800b3c4a Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/dirsnapshot.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/echo.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/echo.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..3ec91817d6156cde18c1329d9440697b5cfd8644 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/echo.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/event_backport.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/event_backport.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..91a371fc3956870909396d058ffc04fb69363b17 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/event_backport.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/importlib2.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/importlib2.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..5d696de5755b068b890ebb596a33ec55808591a6 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/importlib2.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/platform.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/platform.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..fc7186f5f5089990870260463320fc37b5ae37d0 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/platform.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/unicode_paths.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/unicode_paths.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d83a011d541f012ff074594f591c0beff48c55d8 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/unicode_paths.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/win32stat.cpython-312.pyc b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/win32stat.cpython-312.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b41f31c01389b51d925f18003934b828cfd4dbb8 Binary files /dev/null and b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/__pycache__/win32stat.cpython-312.pyc differ diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/bricks.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/bricks.py new file mode 100644 index 0000000000000000000000000000000000000000..b1c2bedc5e0a74cb7c664c3f094666c371f68905 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/bricks.py @@ -0,0 +1,249 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# +# Copyright 2011 Yesudeep Mangalapilly +# Copyright 2012 Google, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + + +""" +Utility collections or "bricks". + +:module: watchdog.utils.bricks +:author: yesudeep@google.com (Yesudeep Mangalapilly) +:author: lalinsky@gmail.com (Lukáš Lalinský) +:author: python@rcn.com (Raymond Hettinger) + +Classes +======= +.. autoclass:: OrderedSetQueue + :members: + :show-inheritance: + :inherited-members: + +.. autoclass:: OrderedSet + +""" + +import sys +from collections.abc import MutableSet +from .compat import queue + +class SkipRepeatsQueue(queue.Queue): + + """Thread-safe implementation of an special queue where a + put of the last-item put'd will be dropped. + + The implementation leverages locking already implemented in the base class + redefining only the primitives. + + Queued items must be immutable and hashable so that they can be used + as dictionary keys. You must implement **only read-only properties** and + the :meth:`Item.__hash__()`, :meth:`Item.__eq__()`, and + :meth:`Item.__ne__()` methods for items to be hashable. + + An example implementation follows:: + + class Item(object): + def __init__(self, a, b): + self._a = a + self._b = b + + @property + def a(self): + return self._a + + @property + def b(self): + return self._b + + def _key(self): + return (self._a, self._b) + + def __eq__(self, item): + return self._key() == item._key() + + def __ne__(self, item): + return self._key() != item._key() + + def __hash__(self): + return hash(self._key()) + + based on the OrderedSetQueue below + """ + + def _init(self, maxsize): + queue.Queue._init(self, maxsize) + self._last_item = None + + def _put(self, item): + if item != self._last_item: + queue.Queue._put(self, item) + self._last_item = item + else: + # `put` increments `unfinished_tasks` even if we did not put + # anything into the queue here + self.unfinished_tasks -= 1 + + def _get(self): + item = queue.Queue._get(self) + if item is self._last_item: + self._last_item = None + return item + + +class OrderedSetQueue(queue.Queue): + + """Thread-safe implementation of an ordered set queue. + + Disallows adding a duplicate item while maintaining the + order of items in the queue. The implementation leverages + locking already implemented in the base class + redefining only the primitives. Since the internal queue + is not replaced, the order is maintained. The set is used + merely to check for the existence of an item. + + Queued items must be immutable and hashable so that they can be used + as dictionary keys. You must implement **only read-only properties** and + the :meth:`Item.__hash__()`, :meth:`Item.__eq__()`, and + :meth:`Item.__ne__()` methods for items to be hashable. + + An example implementation follows:: + + class Item(object): + def __init__(self, a, b): + self._a = a + self._b = b + + @property + def a(self): + return self._a + + @property + def b(self): + return self._b + + def _key(self): + return (self._a, self._b) + + def __eq__(self, item): + return self._key() == item._key() + + def __ne__(self, item): + return self._key() != item._key() + + def __hash__(self): + return hash(self._key()) + + :author: lalinsky@gmail.com (Lukáš Lalinský) + :url: http://stackoverflow.com/questions/1581895/how-check-if-a-task-is-already-in-python-queue + """ + + def _init(self, maxsize): + queue.Queue._init(self, maxsize) + self._set_of_items = set() + + def _put(self, item): + if item not in self._set_of_items: + queue.Queue._put(self, item) + self._set_of_items.add(item) + else: + # `put` increments `unfinished_tasks` even if we did not put + # anything into the queue here + self.unfinished_tasks -= 1 + + def _get(self): + item = queue.Queue._get(self) + self._set_of_items.remove(item) + return item + + +if sys.version_info >= (2, 6, 0): + KEY, PREV, NEXT = list(range(3)) + + class OrderedSet(MutableSet): + + """ + Implementation based on a doubly-linked link and an internal dictionary. + This design gives :class:`OrderedSet` the same big-Oh running times as + regular sets including O(1) adds, removes, and lookups as well as + O(n) iteration. + + .. ADMONITION:: Implementation notes + + Runs on Python 2.6 or later (and runs on Python 3.0 or later + without any modifications). + + :author: python@rcn.com (Raymond Hettinger) + :url: http://code.activestate.com/recipes/576694/ + """ + + def __init__(self, iterable=None): + self.end = end = [] + end += [None, end, end] # sentinel node for doubly linked list + self.map = {} # key --> [key, prev, next] + if iterable is not None: + self |= iterable + + def __len__(self): + return len(self.map) + + def __contains__(self, key): + return key in self.map + + def add(self, key): + if key not in self.map: + end = self.end + curr = end[PREV] + curr[NEXT] = end[PREV] = self.map[key] = [key, curr, end] + + def discard(self, key): + if key in self.map: + key, prev, _next = self.map.pop(key) + prev[NEXT] = _next + _next[PREV] = prev + + def __iter__(self): + end = self.end + curr = end[NEXT] + while curr is not end: + yield curr[KEY] + curr = curr[NEXT] + + def __reversed__(self): + end = self.end + curr = end[PREV] + while curr is not end: + yield curr[KEY] + curr = curr[PREV] + + def pop(self, last=True): + if not self: + raise KeyError('set is empty') + key = next(reversed(self)) if last else next(iter(self)) + self.discard(key) + return key + + def __repr__(self): + if not self: + return '%s()' % (self.__class__.__name__,) + return '%s(%r)' % (self.__class__.__name__, list(self)) + + def __eq__(self, other): + if isinstance(other, OrderedSet): + return len(self) == len(other) and list(self) == list(other) + return set(self) == set(other) + + def __del__(self): + self.clear() # remove circular references diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/compat.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/compat.py new file mode 100644 index 0000000000000000000000000000000000000000..0f6e7947b924a0bd07a99b7e6523c8a161beac42 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/compat.py @@ -0,0 +1,29 @@ +# -*- coding: utf-8 -*- +# +# Copyright 2014 Thomas Amland +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +import sys + +__all__ = ['queue', 'Event'] + +try: + import queue +except ImportError: + import Queue as queue + + +if sys.version_info < (2, 7): + from watchdog.utils.event_backport import Event +else: + from threading import Event \ No newline at end of file diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/decorators.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/decorators.py new file mode 100644 index 0000000000000000000000000000000000000000..abb325c1c1028746cf2a9a1a99bf81432bbae657 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/decorators.py @@ -0,0 +1,198 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# Most of this code was obtained from the Python documentation online. + +"""Decorator utility functions. + +decorators: +- synchronized +- propertyx +- accepts +- returns +- singleton +- attrs +- deprecated +""" + +import functools +import warnings +import threading +import sys + + +def synchronized(lock=None): + """Decorator that synchronizes a method or a function with a mutex lock. + + Example usage: + + @synchronized() + def operation(self, a, b): + ... + """ + if lock is None: + lock = threading.Lock() + + def wrapper(function): + def new_function(*args, **kwargs): + lock.acquire() + try: + return function(*args, **kwargs) + finally: + lock.release() + + return new_function + + return wrapper + + +def propertyx(function): + """Decorator to easily create properties in classes. + + Example: + + class Angle(object): + def __init__(self, rad): + self._rad = rad + + @property + def rad(): + def fget(self): + return self._rad + def fset(self, angle): + if isinstance(angle, Angle): + angle = angle.rad + self._rad = float(angle) + + Arguments: + - `function`: The function to be decorated. + """ + keys = ('fget', 'fset', 'fdel') + func_locals = {'doc': function.__doc__} + + def probe_func(frame, event, arg): + if event == 'return': + locals = frame.f_locals + func_locals.update(dict((k, locals.get(k)) for k in keys)) + sys.settrace(None) + return probe_func + + sys.settrace(probe_func) + function() + return property(**func_locals) + + +def accepts(*types): + """Decorator to ensure that the decorated function accepts the given types as arguments. + + Example: + @accepts(int, (int,float)) + @returns((int,float)) + def func(arg1, arg2): + return arg1 * arg2 + """ + + def check_accepts(f): + assert len(types) == f.__code__.co_argcount + + def new_f(*args, **kwds): + for (a, t) in zip(args, types): + assert isinstance(a, t),\ + "arg %r does not match %s" % (a, t) + return f(*args, **kwds) + + new_f.__name__ = f.__name__ + return new_f + + return check_accepts + + +def returns(rtype): + """Decorator to ensure that the decorated function returns the given + type as argument. + + Example: + @accepts(int, (int,float)) + @returns((int,float)) + def func(arg1, arg2): + return arg1 * arg2 + """ + + def check_returns(f): + def new_f(*args, **kwds): + result = f(*args, **kwds) + assert isinstance(result, rtype),\ + "return value %r does not match %s" % (result, rtype) + return result + + new_f.__name__ = f.__name__ + return new_f + + return check_returns + + +def singleton(cls): + """Decorator to ensures a class follows the singleton pattern. + + Example: + @singleton + class MyClass: + ... + """ + instances = {} + + def getinstance(): + if cls not in instances: + instances[cls] = cls() + return instances[cls] + + return getinstance + + +def attrs(**kwds): + """Decorator to add attributes to a function. + + Example: + + @attrs(versionadded="2.2", + author="Guido van Rossum") + def mymethod(f): + ... + """ + + def decorate(f): + for k in kwds: + setattr(f, k, kwds[k]) + return f + + return decorate + + +def deprecated(func): + """This is a decorator which can be used to mark functions + as deprecated. It will result in a warning being emitted + when the function is used. + + ## Usage examples ## + @deprecated + def my_func(): + pass + + @other_decorators_must_be_upper + @deprecated + def my_func(): + pass + """ + + @functools.wraps(func) + def new_func(*args, **kwargs): + warnings.warn_explicit( + "Call to deprecated function %(funcname)s." % { + 'funcname': func.__name__, + }, + category=DeprecationWarning, + filename=func.__code__.co_filename, + lineno=func.__code__.co_firstlineno + 1 + ) + return func(*args, **kwargs) + + return new_func diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/delayed_queue.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/delayed_queue.py new file mode 100644 index 0000000000000000000000000000000000000000..6d98a50469b4bbe89f20639005fe9ba1e3f8d3cc --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/delayed_queue.py @@ -0,0 +1,88 @@ +# -*- coding: utf-8 -*- +# +# Copyright 2014 Thomas Amland +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import time +import threading +from collections import deque + + +class DelayedQueue(object): + + def __init__(self, delay): + self.delay = delay + self._lock = threading.Lock() + self._not_empty = threading.Condition(self._lock) + self._queue = deque() + self._closed = False + + def put(self, element): + """Add element to queue.""" + self._lock.acquire() + self._queue.append((element, time.time())) + self._not_empty.notify() + self._lock.release() + + def close(self): + """Close queue, indicating no more items will be added.""" + self._closed = True + # Interrupt the blocking _not_empty.wait() call in get + self._not_empty.acquire() + self._not_empty.notify() + self._not_empty.release() + + def get(self): + """Remove and return an element from the queue, or this queue has been + closed raise the Closed exception. + """ + while True: + # wait for element to be added to queue + self._not_empty.acquire() + while len(self._queue) == 0 and not self._closed: + self._not_empty.wait() + + if self._closed: + self._not_empty.release() + return None + head, insert_time = self._queue[0] + self._not_empty.release() + + # wait for delay + time_left = insert_time + self.delay - time.time() + while time_left > 0: + time.sleep(time_left) + time_left = insert_time + self.delay - time.time() + + # return element if it's still in the queue + self._lock.acquire() + try: + if len(self._queue) > 0 and self._queue[0][0] is head: + self._queue.popleft() + return head + finally: + self._lock.release() + + def remove(self, predicate): + """Remove and return the first items for which predicate is True, + ignoring delay.""" + try: + self._lock.acquire() + for i, (elem, t) in enumerate(self._queue): + if predicate(elem): + del self._queue[i] + return elem + finally: + self._lock.release() + return None diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/dirsnapshot.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/dirsnapshot.py new file mode 100644 index 0000000000000000000000000000000000000000..c321d0ffe45c3b943e4649c1dbeec9931bf0db5d --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/dirsnapshot.py @@ -0,0 +1,293 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# +# Copyright 2011 Yesudeep Mangalapilly +# Copyright 2012 Google, Inc. +# Copyright 2014 Thomas Amland +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +""" +:module: watchdog.utils.dirsnapshot +:synopsis: Directory snapshots and comparison. +:author: yesudeep@google.com (Yesudeep Mangalapilly) + +.. ADMONITION:: Where are the moved events? They "disappeared" + + This implementation does not take partition boundaries + into consideration. It will only work when the directory + tree is entirely on the same file system. More specifically, + any part of the code that depends on inode numbers can + break if partition boundaries are crossed. In these cases, + the snapshot diff will represent file/directory movement as + created and deleted events. + +Classes +------- +.. autoclass:: DirectorySnapshot + :members: + :show-inheritance: + +.. autoclass:: DirectorySnapshotDiff + :members: + :show-inheritance: + +""" + +import errno +import os +from stat import S_ISDIR +from wandb_watchdog.utils import stat as default_stat + + +class DirectorySnapshotDiff(object): + """ + Compares two directory snapshots and creates an object that represents + the difference between the two snapshots. + + :param ref: + The reference directory snapshot. + :type ref: + :class:`DirectorySnapshot` + :param snapshot: + The directory snapshot which will be compared + with the reference snapshot. + :type snapshot: + :class:`DirectorySnapshot` + """ + + def __init__(self, ref, snapshot): + created = snapshot.paths - ref.paths + deleted = ref.paths - snapshot.paths + + # check that all unchanged paths have the same inode + for path in ref.paths & snapshot.paths: + if ref.inode(path) != snapshot.inode(path): + created.add(path) + deleted.add(path) + + # find moved paths + moved = set() + for path in set(deleted): + inode = ref.inode(path) + new_path = snapshot.path(inode) + if new_path: + # file is not deleted but moved + deleted.remove(path) + moved.add((path, new_path)) + + for path in set(created): + inode = snapshot.inode(path) + old_path = ref.path(inode) + if old_path: + created.remove(path) + moved.add((old_path, path)) + + # find modified paths + # first check paths that have not moved + modified = set() + for path in ref.paths & snapshot.paths: + if ref.inode(path) == snapshot.inode(path): + if ref.mtime(path) != snapshot.mtime(path): + modified.add(path) + + for (old_path, new_path) in moved: + if ref.mtime(old_path) != snapshot.mtime(new_path): + modified.add(old_path) + + self._dirs_created = [path for path in created if snapshot.isdir(path)] + self._dirs_deleted = [path for path in deleted if ref.isdir(path)] + self._dirs_modified = [path for path in modified if ref.isdir(path)] + self._dirs_moved = [(frm, to) for (frm, to) in moved if ref.isdir(frm)] + + self._files_created = list(created - set(self._dirs_created)) + self._files_deleted = list(deleted - set(self._dirs_deleted)) + self._files_modified = list(modified - set(self._dirs_modified)) + self._files_moved = list(moved - set(self._dirs_moved)) + + @property + def files_created(self): + """List of files that were created.""" + return self._files_created + + @property + def files_deleted(self): + """List of files that were deleted.""" + return self._files_deleted + + @property + def files_modified(self): + """List of files that were modified.""" + return self._files_modified + + @property + def files_moved(self): + """ + List of files that were moved. + + Each event is a two-tuple the first item of which is the path + that has been renamed to the second item in the tuple. + """ + return self._files_moved + + @property + def dirs_modified(self): + """ + List of directories that were modified. + """ + return self._dirs_modified + + @property + def dirs_moved(self): + """ + List of directories that were moved. + + Each event is a two-tuple the first item of which is the path + that has been renamed to the second item in the tuple. + """ + return self._dirs_moved + + @property + def dirs_deleted(self): + """ + List of directories that were deleted. + """ + return self._dirs_deleted + + @property + def dirs_created(self): + """ + List of directories that were created. + """ + return self._dirs_created + +class DirectorySnapshot(object): + """ + A snapshot of stat information of files in a directory. + + :param path: + The directory path for which a snapshot should be taken. + :type path: + ``str`` + :param recursive: + ``True`` if the entire directory tree should be included in the + snapshot; ``False`` otherwise. + :type recursive: + ``bool`` + :param walker_callback: + .. deprecated:: 0.7.2 + :param stat: + Use custom stat function that returns a stat structure for path. + Currently only st_dev, st_ino, st_mode and st_mtime are needed. + + A function with the signature ``walker_callback(path, stat_info)`` + which will be called for every entry in the directory tree. + :param listdir: + Use custom listdir function. See ``os.listdir`` for details. + """ + + def __init__(self, path, recursive=True, + walker_callback=(lambda p, s: None), + stat=default_stat, + listdir=os.listdir): + self._stat_info = {} + self._inode_to_path = {} + + st = stat(path) + self._stat_info[path] = st + self._inode_to_path[(st.st_ino, st.st_dev)] = path + + def walk(root): + try: + paths = [os.path.join(root, name) for name in listdir(root)] + except OSError as e: + # Directory may have been deleted between finding it in the directory + # list of its parent and trying to delete its contents. If this + # happens we treat it as empty. + if e.errno == errno.ENOENT: + return + else: + raise + entries = [] + for p in paths: + try: + entries.append((p, stat(p))) + except OSError: + continue + for _ in entries: + yield _ + if recursive: + for path, st in entries: + if S_ISDIR(st.st_mode): + for _ in walk(path): + yield _ + + for p, st in walk(path): + i = (st.st_ino, st.st_dev) + self._inode_to_path[i] = p + self._stat_info[p] = st + walker_callback(p, st) + + @property + def paths(self): + """ + Set of file/directory paths in the snapshot. + """ + return set(self._stat_info.keys()) + + def path(self, id): + """ + Returns path for id. None if id is unknown to this snapshot. + """ + return self._inode_to_path.get(id) + + def inode(self, path): + """ Returns an id for path. """ + st = self._stat_info[path] + return (st.st_ino, st.st_dev) + + def isdir(self, path): + return S_ISDIR(self._stat_info[path].st_mode) + + def mtime(self, path): + return self._stat_info[path].st_mtime + + def stat_info(self, path): + """ + Returns a stat information object for the specified path from + the snapshot. + + Attached information is subject to change. Do not use unless + you specify `stat` in constructor. Use :func:`inode`, :func:`mtime`, + :func:`isdir` instead. + + :param path: + The path for which stat information should be obtained + from a snapshot. + """ + return self._stat_info[path] + + def __sub__(self, previous_dirsnap): + """Allow subtracting a DirectorySnapshot object instance from + another. + + :returns: + A :class:`DirectorySnapshotDiff` object. + """ + return DirectorySnapshotDiff(previous_dirsnap, self) + + def __str__(self): + return self.__repr__() + + def __repr__(self): + return str(self._stat_info) diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/echo.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/echo.py new file mode 100644 index 0000000000000000000000000000000000000000..12803e030d5c75308657c5f6717f76059907774c --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/echo.py @@ -0,0 +1,157 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# echo.py: Tracing function calls using Python decorators. +# +# Written by Thomas Guest +# Please see http://wordaligned.org/articles/echo +# +# Place into the public domain. + +""" Echo calls made to functions and methods in a module. + +"Echoing" a function call means printing out the name of the function +and the values of its arguments before making the call (which is more +commonly referred to as "tracing", but Python already has a trace module). + +Example: to echo calls made to functions in "my_module" do: + + import echo + import my_module + echo.echo_module(my_module) + +Example: to echo calls made to functions in "my_module.my_class" do: + + echo.echo_class(my_module.my_class) + +Alternatively, echo.echo can be used to decorate functions. Calls to the +decorated function will be echoed. + +Example: + + @echo.echo + def my_function(args): + pass +""" +import inspect +import sys + + +def name(item): + " Return an item's name. " + return item.__name__ + + +def is_classmethod(instancemethod, klass): + " Determine if an instancemethod is a classmethod. " + return inspect.ismethod(instancemethod) and instancemethod.__self__ is klass + +def is_static_method(method, klass): + """Returns True if method is an instance method of klass.""" + for c in klass.mro(): + if name(method) in c.__dict__: + return isinstance(c.__dict__[name(method)], staticmethod) + else: + return False + +def is_class_private_name(name): + " Determine if a name is a class private name. " + # Exclude system defined names such as __init__, __add__ etc + return name.startswith("__") and not name.endswith("__") + + +def method_name(method): + """ Return a method's name. + + This function returns the name the method is accessed by from + outside the class (i.e. it prefixes "private" methods appropriately). + """ + mname = name(method) + if is_class_private_name(mname): + mname = "_%s%s" % (name(method.__self__.__class__), mname) + return mname + + +def format_arg_value(arg_val): + """ Return a string representing a (name, value) pair. + + >>> format_arg_value(('x', (1, 2, 3))) + 'x=(1, 2, 3)' + """ + arg, val = arg_val + return "%s=%r" % (arg, val) + + +def echo(fn, write=sys.stdout.write): + """ Echo calls to a function. + + Returns a decorated version of the input function which "echoes" calls + made to it by writing out the function's name and the arguments it was + called with. + """ + import functools + # Unpack function's arg count, arg names, arg defaults + code = fn.__code__ + argcount = code.co_argcount + argnames = code.co_varnames[:argcount] + fn_defaults = fn.__defaults__ or list() + argdefs = dict(list(zip(argnames[-len(fn_defaults):], fn_defaults))) + + @functools.wraps(fn) + def wrapped(*v, **k): + # Collect function arguments by chaining together positional, + # defaulted, extra positional and keyword arguments. + positional = list(map(format_arg_value, list(zip(argnames, v)))) + defaulted = [format_arg_value((a, argdefs[a])) + for a in argnames[len(v):] if a not in k] + nameless = list(map(repr, v[argcount:])) + keyword = list(map(format_arg_value, list(k.items()))) + args = positional + defaulted + nameless + keyword + write("%s(%s)\n" % (name(fn), ", ".join(args))) + return fn(*v, **k) + + return wrapped + + +def echo_instancemethod(klass, method, write=sys.stdout.write): + """ Change an instancemethod so that calls to it are echoed. + + Replacing a classmethod is a little more tricky. + See: http://www.python.org/doc/current/ref/types.html + """ + mname = method_name(method) + never_echo = "__str__", "__repr__", # Avoid recursion printing method calls + if mname in never_echo: + pass + elif is_classmethod(method, klass): + setattr(klass, mname, classmethod(echo(method.__func__, write))) + else: + setattr(klass, mname, echo(method, write)) + +def echo_class(klass, write=sys.stdout.write): + """ Echo calls to class methods and static functions + """ + for _, method in inspect.getmembers(klass, inspect.ismethod): + #In python 3 only class methods are returned here, but in python2 instance methods are too. + echo_instancemethod(klass, method, write) + for _, fn in inspect.getmembers(klass, inspect.isfunction): + if is_static_method(fn, klass): + setattr(klass, name(fn), staticmethod(echo(fn, write))) + else: + #It's not a class or a static method, so it must be an instance method. + #This should only be called in python 3, because in python 3 instance methods are considered functions. + echo_instancemethod(klass, fn, write) + +def echo_module(mod, write=sys.stdout.write): + """ Echo calls to functions and methods in a module. + """ + for fname, fn in inspect.getmembers(mod, inspect.isfunction): + setattr(mod, fname, echo(fn, write)) + for _, klass in inspect.getmembers(mod, inspect.isclass): + echo_class(klass, write) + +if __name__ == "__main__": + import doctest + + optionflags = doctest.ELLIPSIS + doctest.testfile('echoexample.txt', optionflags=optionflags) + doctest.testmod(optionflags=optionflags) diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/event_backport.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/event_backport.py new file mode 100644 index 0000000000000000000000000000000000000000..5c136e46d54839347c36e7f3ff81ff64f51b0d5b --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/event_backport.py @@ -0,0 +1,41 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# Backport of Event from py2.7 (method wait in py2.6 returns None) + +from threading import Condition, Lock + + +class Event(object): + + def __init__(self,): + self.__cond = Condition(Lock()) + self.__flag = False + + def isSet(self): + return self.__flag + + is_set = isSet + + def set(self): + self.__cond.acquire() + try: + self.__flag = True + self.__cond.notify_all() + finally: + self.__cond.release() + + def clear(self): + self.__cond.acquire() + try: + self.__flag = False + finally: + self.__cond.release() + + def wait(self, timeout=None): + self.__cond.acquire() + try: + if not self.__flag: + self.__cond.wait(timeout) + return self.__flag + finally: + self.__cond.release() diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/importlib2.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/importlib2.py new file mode 100644 index 0000000000000000000000000000000000000000..5ad3ec57204c6b8a6db0589b631c849f0b145a00 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/importlib2.py @@ -0,0 +1,40 @@ +# The MIT License (MIT) + +# Copyright (c) 2013 Peter M. Elias + +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: + +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. + +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE + + +def import_module(target, relative_to=None): + target_parts = target.split('.') + target_depth = target_parts.count('') + target_path = target_parts[target_depth:] + target = target[target_depth:] + fromlist = [target] + if target_depth and relative_to: + relative_parts = relative_to.split('.') + relative_to = '.'.join(relative_parts[:-(target_depth - 1) or None]) + if len(target_path) > 1: + relative_to = '.'.join(filter(None, [relative_to]) + target_path[:-1]) + fromlist = target_path[-1:] + target = fromlist[0] + elif not relative_to: + fromlist = [] + mod = __import__(relative_to or target, globals(), locals(), fromlist) + return getattr(mod, target, mod) diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/platform.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/platform.py new file mode 100644 index 0000000000000000000000000000000000000000..239c6a25829dde1b70cfab092678820ca91a326d --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/platform.py @@ -0,0 +1,57 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# +# Copyright 2011 Yesudeep Mangalapilly +# Copyright 2012 Google, Inc. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + + +import sys + +PLATFORM_WINDOWS = 'windows' +PLATFORM_LINUX = 'linux' +PLATFORM_BSD = 'bsd' +PLATFORM_DARWIN = 'darwin' +PLATFORM_UNKNOWN = 'unknown' + + +def get_platform_name(): + if sys.platform.startswith("win"): + return PLATFORM_WINDOWS + elif sys.platform.startswith('darwin'): + return PLATFORM_DARWIN + elif sys.platform.startswith('linux'): + return PLATFORM_LINUX + elif sys.platform.startswith(('dragonfly', 'freebsd', 'netbsd', 'openbsd', )): + return PLATFORM_BSD + else: + return PLATFORM_UNKNOWN + +__platform__ = get_platform_name() + + +def is_linux(): + return __platform__ == PLATFORM_LINUX + + +def is_bsd(): + return __platform__ == PLATFORM_BSD + + +def is_darwin(): + return __platform__ == PLATFORM_DARWIN + + +def is_windows(): + return __platform__ == PLATFORM_WINDOWS diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/unicode_paths.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/unicode_paths.py new file mode 100644 index 0000000000000000000000000000000000000000..9e4b4b425fd9c7420d2a01ff5ddf0c7eadc60b06 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/unicode_paths.py @@ -0,0 +1,64 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# +# Copyright (c) 2013 Will Bond +# +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in +# all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + + +import sys + +from wandb_watchdog.utils import platform + +try: + # Python 2 + str_cls = unicode + bytes_cls = str +except NameError: + # Python 3 + str_cls = str + bytes_cls = bytes + + +# This is used by Linux when the locale seems to be improperly set. UTF-8 tends +# to be the encoding used by all distros, so this is a good fallback. +fs_fallback_encoding = 'utf-8' +fs_encoding = sys.getfilesystemencoding() or fs_fallback_encoding + + +def encode(path): + if isinstance(path, str_cls): + try: + path = path.encode(fs_encoding, 'strict') + except UnicodeEncodeError: + if not platform.is_linux(): + raise + path = path.encode(fs_fallback_encoding, 'strict') + return path + + +def decode(path): + if isinstance(path, bytes_cls): + try: + path = path.decode(fs_encoding, 'strict') + except UnicodeDecodeError: + if not platform.is_linux(): + raise + path = path.decode(fs_fallback_encoding, 'strict') + return path diff --git a/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/win32stat.py b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/win32stat.py new file mode 100644 index 0000000000000000000000000000000000000000..398d1067942ab987d230ec87eaa54a3f58fd67c5 --- /dev/null +++ b/lib/python3.12/site-packages/wandb/vendor/watchdog_0_9_0/wandb_watchdog/utils/win32stat.py @@ -0,0 +1,123 @@ +# -*- coding: utf-8 -*- +# +# Copyright 2014 Thomas Amland +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +""" +:module: watchdog.utils.win32stat +:synopsis: Implementation of stat with st_ino and st_dev support. + +Functions +--------- + +.. autofunction:: stat + +""" + +import ctypes +import ctypes.wintypes +import stat as stdstat +from collections import namedtuple + + +INVALID_HANDLE_VALUE = ctypes.c_void_p(-1).value +OPEN_EXISTING = 3 +FILE_READ_ATTRIBUTES = 0x80 +FILE_ATTRIBUTE_NORMAL = 0x80 +FILE_ATTRIBUTE_READONLY = 0x1 +FILE_ATTRIBUTE_DIRECTORY = 0x10 +FILE_FLAG_BACKUP_SEMANTICS = 0x02000000 +FILE_FLAG_OPEN_REPARSE_POINT = 0x00200000 + + +class FILETIME(ctypes.Structure): + _fields_ = [("dwLowDateTime", ctypes.wintypes.DWORD), + ("dwHighDateTime", ctypes.wintypes.DWORD)] + + +class BY_HANDLE_FILE_INFORMATION(ctypes.Structure): + _fields_ = [('dwFileAttributes', ctypes.wintypes.DWORD), + ('ftCreationTime', FILETIME), + ('ftLastAccessTime', FILETIME), + ('ftLastWriteTime', FILETIME), + ('dwVolumeSerialNumber', ctypes.wintypes.DWORD), + ('nFileSizeHigh', ctypes.wintypes.DWORD), + ('nFileSizeLow', ctypes.wintypes.DWORD), + ('nNumberOfLinks', ctypes.wintypes.DWORD), + ('nFileIndexHigh', ctypes.wintypes.DWORD), + ('nFileIndexLow', ctypes.wintypes.DWORD)] + + +CreateFile = ctypes.windll.kernel32.CreateFileW +CreateFile.restype = ctypes.wintypes.HANDLE +CreateFile.argtypes = ( + ctypes.c_wchar_p, + ctypes.wintypes.DWORD, + ctypes.wintypes.DWORD, + ctypes.c_void_p, + ctypes.wintypes.DWORD, + ctypes.wintypes.DWORD, + ctypes.wintypes.HANDLE, +) + +GetFileInformationByHandle = ctypes.windll.kernel32.GetFileInformationByHandle +GetFileInformationByHandle.restype = ctypes.wintypes.BOOL +GetFileInformationByHandle.argtypes = ( + ctypes.wintypes.HANDLE, + ctypes.wintypes.POINTER(BY_HANDLE_FILE_INFORMATION), +) + +CloseHandle = ctypes.windll.kernel32.CloseHandle +CloseHandle.restype = ctypes.wintypes.BOOL +CloseHandle.argtypes = (ctypes.wintypes.HANDLE,) + + +StatResult = namedtuple('StatResult', 'st_dev st_ino st_mode st_mtime') + +def _to_mode(attr): + m = 0 + if (attr & FILE_ATTRIBUTE_DIRECTORY): + m |= stdstat.S_IFDIR | 0o111 + else: + m |= stdstat.S_IFREG + if (attr & FILE_ATTRIBUTE_READONLY): + m |= 0o444 + else: + m |= 0o666 + return m + +def _to_unix_time(ft): + t = (ft.dwHighDateTime) << 32 | ft.dwLowDateTime + return (t / 10000000) - 11644473600 + +def stat(path): + hfile = CreateFile(path, + FILE_READ_ATTRIBUTES, + 0, + None, + OPEN_EXISTING, + FILE_ATTRIBUTE_NORMAL | FILE_FLAG_BACKUP_SEMANTICS | FILE_FLAG_OPEN_REPARSE_POINT, + None) + if hfile == INVALID_HANDLE_VALUE: + raise ctypes.WinError + info = BY_HANDLE_FILE_INFORMATION() + r = GetFileInformationByHandle(hfile, info) + CloseHandle(hfile) + if not r: + raise ctypes.WinError + return StatResult(st_dev=info.dwVolumeSerialNumber, + st_ino=(info.nFileIndexHigh << 32) + info.nFileIndexLow, + st_mode=_to_mode(info.dwFileAttributes), + st_mtime=_to_unix_time(info.ftLastWriteTime) + )