diff --git a/docs/source/en/_toctree.yml b/docs/source/en/_toctree.yml index 4b4e4247195d..3abfb9052cd2 100644 --- a/docs/source/en/_toctree.yml +++ b/docs/source/en/_toctree.yml @@ -114,6 +114,8 @@ title: Intel Gaudi - local: optimization/neuron title: AWS Neuron + - local: optimization/tpu + title: TPU title: Hardware-specific acceleration - isExpanded: false sections: diff --git a/docs/source/en/optimization/tpu.md b/docs/source/en/optimization/tpu.md new file mode 100644 index 000000000000..619e84f3a447 --- /dev/null +++ b/docs/source/en/optimization/tpu.md @@ -0,0 +1,151 @@ + + +# TorchTPU + +[TorchTPU](https://github.com/google-pytorch/torch_tpu/) is a PyTorch backend for Google's Tensor Processing Units (TPUs), which lets you run Diffusers pipelines on Cloud TPUs (v6e, v5p, etc.) with minimal code changes. + +Two execution modes are available: + +| Mode | Constant | How to activate | Notes | +|---|---|---|---| +| Strict eager (default) | `EagerMode.DEFER_NEVER` | `pipe.to("tpu")` | Operations dispatched one at a time, asynchronous | +| Compile | — | `torch.compile(module, backend="tpu")` | AOT compilation with `TpuBackend` | + +Follow the [TorchTPU installation guide](https://github.com/google-pytorch/torch_tpu/). Once installed, `import torch` +loads it automatically and registers the `"tpu"` device, so `pipe.to("tpu")` is the only change needed. Add +`import torch_tpu` only if you disabled backend autoloading with `TORCH_DEVICE_BACKEND_AUTOLOAD=0`. + +## Eager mode + +FLUX.1-schnell doesn't fit on a single v6e chip all at once, so use [`~DiffusionPipeline.enable_model_cpu_offload`] to +move each model to the TPU only while it runs. It detects the `"tpu"` device automatically. + +```python +import torch + +from diffusers import FluxPipeline + +pipe = FluxPipeline.from_pretrained("black-forest-labs/FLUX.1-schnell", dtype=torch.bfloat16) +pipe.enable_model_cpu_offload() + +image = pipe( + prompt="a golden retriever surfing a wave, photorealistic", + height=1024, + width=1024, + num_inference_steps=4, + guidance_scale=0.0, +).images[0] + +image.save("output.png") +``` + +If a model is too large for a single chip, or you have several chips and want lower latency, shard the models +across chips instead. See the [Tensor parallelism](#tensor-parallelism) section. + +## Compiled mode + +TorchTPU registers `"tpu"` as a `torch.compile` backend name (`TpuBackend` under the hood), so +components compile like any other `torch.compile` target. The first +call (warmup) is slow because it compiles; later calls with the same shapes reuse the compiled graph. + +> [!IMPORTANT] +> TorchTPU requires **static shapes**, so pass `dynamic=False`. A new `height` or `width` compiles again for that +> shape, once; shapes already seen are reused. Changing `num_inference_steps` doesn't recompile. + +When the whole pipeline fits on one chip, move it to the TPU and compile the full transformer. Stable Diffusion 3.5 +Medium (~15GB in bf16) fits on a single v6e chip. + +```python +import torch + +from diffusers import StableDiffusion3Pipeline + +pipe = StableDiffusion3Pipeline.from_pretrained("stabilityai/stable-diffusion-3.5-medium", dtype=torch.bfloat16) +pipe.to("tpu") +pipe.transformer.compile(backend="tpu", fullgraph=True, dynamic=False) + +# Warmup — triggers static graph compilation. +pipe(prompt="warmup", height=1024, width=1024, num_inference_steps=40, guidance_scale=4.5) + +# Later calls with the same shapes reuse the compiled graph. +image = pipe( + prompt="a golden retriever surfing a wave, photorealistic", + height=1024, + width=1024, + num_inference_steps=40, + guidance_scale=4.5, +).images[0] + +image.save("output.png") +``` + +If the pipeline doesn't fit on one chip: + +- With several chips, shard it with [tensor parallelism](#tensor-parallelism) instead. Everything stays on the TPU, + and `pipe.transformer.compile(...)` works the same way on the sharded transformer. +- With [`~DiffusionPipeline.enable_model_cpu_offload`], the offload hooks can't be traced by `torch.compile`. Compile + only the transformer's repeated blocks instead, with + `pipe.transformer.compile_repeated_blocks(backend="tpu", fullgraph=True, dynamic=False)`. + +## Tensor parallelism + +Shard models too large for one chip across several. FLUX.2-dev's text encoder (~48GB) and transformer (~64GB) each +exceed a single chip, so the example below shards both: + +- the transformer with [`TensorParallelConfig`], passed to the `parallel_config` argument of [`~ModelMixin.from_pretrained`]. Each rank reads only its own slice of every sharded weight, so the full model is never materialized. For general TP details (`_tp_plan`, colwise/rowwise), see the [Tensor parallelism](../training/distributed_inference#tensor-parallelism) guide. +- the text encoder with Transformers' own [tensor parallelism](https://huggingface.co/docs/transformers/perf_infer_gpu_multi), passing a [`~transformers.DistributedConfig`] and the same mesh. + +On TPU, initialize the process group with `backend="tpu_dist"` and build the mesh with `DeviceMesh("tpu", ...)`. + +```python +import torch +import torch.distributed as dist +from torch.distributed.device_mesh import DeviceMesh +from transformers import DistributedConfig, Mistral3ForConditionalGeneration + +from diffusers import Flux2Pipeline, Flux2Transformer2DModel, TensorParallelConfig + +dist.init_process_group(backend="tpu_dist") +mesh = DeviceMesh("tpu", list(range(dist.get_world_size()))) + +repo_id = "black-forest-labs/FLUX.2-dev" +text_encoder = Mistral3ForConditionalGeneration.from_pretrained( + repo_id, + subfolder="text_encoder", + dtype=torch.bfloat16, + distributed_config=DistributedConfig(tp_plan="auto"), + device_mesh=mesh, +) +transformer = Flux2Transformer2DModel.from_pretrained( + repo_id, subfolder="transformer", dtype=torch.bfloat16, parallel_config=TensorParallelConfig(mesh=mesh) +) +pipe = Flux2Pipeline.from_pretrained( + repo_id, text_encoder=text_encoder, transformer=transformer, dtype=torch.bfloat16 +) +pipe.vae.to("tpu") + +image = pipe( + prompt="a golden retriever surfing a wave, photorealistic", + num_inference_steps=28, + generator=torch.Generator("cpu").manual_seed(0), +).images[0] +if dist.get_rank() == 0: + image.save("output.png") +``` + +Launch one process per chip. Set `--nproc_per_node` to use all the number of TPU chips on your host. + +```bash +eval $(python -m torch_tpu._internal.distributed.launchers.singlehost_wrapper | sed 's/^/export /') +torchrun --nproc_per_node=8 flux2_tp.py +``` diff --git a/src/diffusers/hooks/tensor_parallel.py b/src/diffusers/hooks/tensor_parallel.py index 0f56c3644ec2..ce0c76f97081 100644 --- a/src/diffusers/hooks/tensor_parallel.py +++ b/src/diffusers/hooks/tensor_parallel.py @@ -22,7 +22,7 @@ logger = get_logger(__name__) # pylint: disable=invalid-name -_SUPPORTED_TP_DEVICES = ("cuda", "neuron") +_SUPPORTED_TP_DEVICES = ("cuda", "neuron", "tpu") class PackedColwiseParallel: diff --git a/src/diffusers/models/_modeling_parallel.py b/src/diffusers/models/_modeling_parallel.py index b54e86d6b4f2..58cbc822f8bf 100644 --- a/src/diffusers/models/_modeling_parallel.py +++ b/src/diffusers/models/_modeling_parallel.py @@ -161,7 +161,7 @@ class TensorParallelConfig: Tensor parallelism shards weight matrices (column-wise and row-wise) across devices. Each device computes a partial result; an AllReduce/AllGather at layer boundaries reconstructs the full output. Uses `torch.distributed.tensor.parallelize_module` with `ColwiseParallel` / `RowwiseParallel` sharding styles. Supported - device types are `"cuda"` and `"neuron"`. + device types are `"cuda"`, `"neuron"` and `"tpu"`. Args: tp_degree (`int`, defaults to `1`): diff --git a/src/diffusers/pipelines/stable_diffusion_3/pipeline_stable_diffusion_3.py b/src/diffusers/pipelines/stable_diffusion_3/pipeline_stable_diffusion_3.py index 9509adde741b..36963b1440f9 100644 --- a/src/diffusers/pipelines/stable_diffusion_3/pipeline_stable_diffusion_3.py +++ b/src/diffusers/pipelines/stable_diffusion_3/pipeline_stable_diffusion_3.py @@ -1066,7 +1066,8 @@ def __call__( # expand the latents if we are doing classifier free guidance latent_model_input = torch.cat([latents] * 2) if self.do_classifier_free_guidance else latents # broadcast to batch dimension in a way that's compatible with ONNX/Core ML - timestep = t.expand(latent_model_input.shape[0]) + # `clone()` so it isn't a view into `timesteps`, which would recompile `torch.compile` every step + timestep = t.expand(latent_model_input.shape[0]).clone() noise_pred = self.transformer( hidden_states=latent_model_input, @@ -1088,7 +1089,7 @@ def __call__( else False ) if skip_guidance_layers is not None and should_skip_layers: - timestep = t.expand(latents.shape[0]) + timestep = t.expand(latents.shape[0]).clone() latent_model_input = latents noise_pred_skip_layers = self.transformer( hidden_states=latent_model_input, diff --git a/src/diffusers/utils/__init__.py b/src/diffusers/utils/__init__.py index b3051dfcb9d1..5f89eb318d92 100644 --- a/src/diffusers/utils/__init__.py +++ b/src/diffusers/utils/__init__.py @@ -116,6 +116,7 @@ is_torch_mlu_available, is_torch_neuronx_available, is_torch_npu_available, + is_torch_tpu_available, is_torch_version, is_torch_xla_available, is_torch_xla_version, diff --git a/src/diffusers/utils/import_utils.py b/src/diffusers/utils/import_utils.py index d2cf394cd9a7..af765021f409 100644 --- a/src/diffusers/utils/import_utils.py +++ b/src/diffusers/utils/import_utils.py @@ -178,6 +178,7 @@ def _is_package_available(pkg_name: str, get_dist_name: bool = False) -> tuple[b _torch_xla_available, _torch_xla_version = _is_package_available("torch_xla") _torch_npu_available, _torch_npu_version = _is_package_available("torch_npu") _torch_mlu_available, _torch_mlu_version = _is_package_available("torch_mlu") +_torch_tpu_available, _torch_tpu_version = _is_package_available("torch_tpu") _torch_neuronx_available, _torch_neuronx_version = _is_package_available("torch_neuronx") _transformers_available, _transformers_version = _is_package_available("transformers") _hf_hub_available, _hf_hub_version = _is_package_available("huggingface_hub") @@ -238,6 +239,10 @@ def is_torch_mlu_available(): return _torch_mlu_available +def is_torch_tpu_available(): + return _torch_tpu_available + + def is_torch_neuronx_available(): return _torch_neuronx_available diff --git a/tests/models/testing_utils/__init__.py b/tests/models/testing_utils/__init__.py index 2d7d5ae23257..0932e46f9b93 100644 --- a/tests/models/testing_utils/__init__.py +++ b/tests/models/testing_utils/__init__.py @@ -23,6 +23,7 @@ ContextParallelAttentionBackendsTesterMixin, ContextParallelTesterMixin, TensorParallelTesterMixin, + TensorParallelTPUTesterMixin, ) from .quantization import ( AutoRoundCompileTesterMixin, @@ -67,6 +68,7 @@ "ContextParallelTesterMixin", "ContextParallelAttentionBackendsTesterMixin", "TensorParallelTesterMixin", + "TensorParallelTPUTesterMixin", "CPUOffloadTesterMixin", "FasterCacheConfigMixin", "FasterCacheTesterMixin", diff --git a/tests/models/testing_utils/parallelism.py b/tests/models/testing_utils/parallelism.py index b525637c953e..eca6bff5d271 100644 --- a/tests/models/testing_utils/parallelism.py +++ b/tests/models/testing_utils/parallelism.py @@ -31,6 +31,7 @@ is_kernels_available, is_tensor_parallel, require_torch_multi_accelerator, + require_torch_tpu, torch_device, ) from .common import calculate_expected_num_shards, compute_module_persistent_sizes @@ -41,6 +42,7 @@ DEVICE_CONFIG = { "cuda": {"backend": "nccl", "module": torch.cuda}, "xpu": {"backend": "xccl", "module": torch.xpu}, + "tpu": {"backend": "tpu_dist", "module": None}, } @@ -245,7 +247,15 @@ def _custom_mesh_worker( def _tensor_parallel_worker( - rank, world_size, master_port, model_class, init_dict, inputs_dict, return_dict, state_dict + rank, + world_size, + master_port, + model_class, + init_dict, + inputs_dict, + return_dict, + state_dict, + device_type=torch_device, ): """Worker function for tensor parallel inference testing. @@ -258,16 +268,19 @@ def _tensor_parallel_worker( os.environ["MASTER_ADDR"] = "localhost" os.environ["MASTER_PORT"] = str(master_port) os.environ["RANK"] = str(rank) + os.environ["LOCAL_RANK"] = str(rank) os.environ["WORLD_SIZE"] = str(world_size) - device_config = DEVICE_CONFIG.get(torch_device, DEVICE_CONFIG["cuda"]) - backend = device_config["backend"] - device_module = device_config["module"] - - dist.init_process_group(backend=backend, rank=rank, world_size=world_size) + device_config = DEVICE_CONFIG.get(device_type, DEVICE_CONFIG["cuda"]) + dist.init_process_group(backend=device_config["backend"], rank=rank, world_size=world_size) - device_module.set_device(rank) - device = torch.device(f"{torch_device}:{rank}") + if device_type == "tpu": + # Each spawned process is bound to one chip. Avoid bf16 matmuls to keep the tolerance tight. + device = torch.device("tpu") + torch.set_float32_matmul_precision("highest") + else: + device_config["module"].set_device(rank) + device = torch.device(f"{device_type}:{rank}") model = model_class(**init_dict) model.load_state_dict(state_dict) @@ -303,8 +316,8 @@ def _tensor_parallel_from_pretrained_worker( """Worker for `from_pretrained(..., parallel_config=...)`, i.e. sharding while reading the checkpoint. Each rank loads only its own slice of every `_tp_plan` parameter straight into a `DTensor` and runs a forward - pass. Rank 0 reports its output and the local/global shapes of one sharded weight so the caller can check both the - numerics and that sharding actually happened. + pass. Rank 0 checks that exactly the parameters `_tp_plan` covers were loaded as `DTensor`s, each with the + placement and local shape its shard spec implies, and reports its output so the caller can check the numerics. """ try: os.environ["MASTER_ADDR"] = "localhost" @@ -316,7 +329,9 @@ def _tensor_parallel_from_pretrained_worker( dist.init_process_group(backend=device_config["backend"], rank=rank, world_size=world_size) device_config["module"].set_device(rank) - from torch.distributed.tensor import DTensor + from torch.distributed.tensor import DTensor, Replicate, Shard + + from diffusers.hooks.tensor_parallel import resolve_tp_shard_specs model = model_class.from_pretrained( checkpoint_dir, parallel_config=TensorParallelConfig(tp_degree=world_size) @@ -330,12 +345,30 @@ def _tensor_parallel_from_pretrained_worker( output = output.full_tensor() if rank == 0: - sharded = {k: v for k, v in model.state_dict().items() if isinstance(v, DTensor)} - assert sharded, "No parameter was sharded into a DTensor by the streaming load." - name, param = next(iter(sharded.items())) + specs = resolve_tp_shard_specs(model, model_class._tp_plan, world_size) + state_dict = model.state_dict() + + # The numerics check alone wouldn't catch a planned parameter left unsharded. + for name, spec in specs.items(): + param = state_dict[name] + assert isinstance(param, DTensor), ( + f"'{name}' is covered by `_tp_plan` but was not loaded as a DTensor." + ) + placement = Replicate() if spec.dim is None else Shard(spec.dim) + assert param.placements == (placement,), ( + f"'{name}' has placements {param.placements}, not {placement}." + ) + expected_local_shape = list(param.shape) + if spec.dim is not None: + expected_local_shape[spec.dim] //= world_size + assert list(param.to_local().shape) == expected_local_shape, ( + f"'{name}' has local shape {list(param.to_local().shape)}, not {expected_local_shape}." + ) + + unplanned = sorted(k for k, v in state_dict.items() if isinstance(v, DTensor) and k not in specs) + assert not unplanned, f"Parameters not covered by `_tp_plan` were loaded as DTensors: {unplanned}" + return_dict["status"] = "success" - return_dict["num_sharded"] = len(sharded) - return_dict["shard_example"] = (name, list(param.to_local().shape), list(param.shape)) return_dict["output"] = output.float().cpu().tolist() except Exception as e: @@ -456,15 +489,75 @@ def test_tensor_parallel_from_pretrained(self, tmp_path, sharded): f"Tensor parallel `from_pretrained` failed: {return_dict.get('error', 'Unknown error')}" ) - name, local_shape, global_shape = return_dict["shard_example"] - assert local_shape != global_shape, ( - f"'{name}' has local shape {local_shape} equal to its global shape, so it was not sharded." - ) - # Sharded matmuls + all-reduce reorder the summation, so allow a small tolerance over the reference. torch.testing.assert_close(reference, torch.tensor(return_dict["output"]), atol=1e-3, rtol=1e-3) +@is_tensor_parallel +@require_torch_tpu +class TensorParallelTPUTesterMixin: + """Same check as `TensorParallelTesterMixin`, spawning one process per TPU chip. + + Run these tests in their own pytest process (e.g. `-k TensorParallelTPU`): `torch.manual_seed` or a backward pass + in an earlier test binds every TPU chip to the pytest process, and the spawned workers then fail. + """ + + # Smallest slice every TPU generation supports, so the test is the same on any CI host. + tp_world_size = 4 + + def test_tensor_parallel_tpu_inference(self, atol=1e-3, rtol=1e-3): + from torch_tpu._internal.distributed.launchers.singlehost_wrapper import prepare_tpu_environment + from torch_tpu._internal.utils import hardware + + if getattr(self.model_class, "_tp_plan", None) is None: + pytest.skip("Model does not define a `_tp_plan` for tensor parallel inference.") + + world_size = self.tp_world_size + if hardware.get_tpu_device_count() < world_size: + pytest.skip(f"Needs at least {world_size} TPU chips.") + init_dict = self.get_init_dict() + num_heads = init_dict.get("num_attention_heads") + if num_heads is not None and num_heads % world_size != 0: + pytest.skip(f"`num_attention_heads` ({num_heads}) is not divisible by tp_degree ({world_size}).") + + # Reference on CPU: touching the TPU here would bind every chip to this process. + inputs_dict = self.get_dummy_inputs(device="cpu") + model = self.model_class(**init_dict).eval() + with torch.no_grad(): + ref_output = model(**inputs_dict, return_dict=False)[0].float() + + tpu_env = ("TORCH_TPU_TOPOLOGY", "TORCH_TPU_SLICEBUILDER_ADDRESSES") + for key in tpu_env: + os.environ.pop(key, None) + prepare_tpu_environment(world_size) + return_dict = mp.Manager().dict() + try: + mp.spawn( + _tensor_parallel_worker, + args=( + world_size, + _find_free_port(), + self.model_class, + init_dict, + inputs_dict, + return_dict, + model.state_dict(), + "tpu", + ), + nprocs=world_size, + join=True, + ) + finally: + for key in tpu_env: + os.environ.pop(key, None) + + assert return_dict.get("status") == "success", ( + f"Tensor parallel inference failed: {return_dict.get('error', 'Unknown error')}" + ) + tp_output = torch.tensor(return_dict["output"]) + torch.testing.assert_close(ref_output, tp_output, atol=atol, rtol=rtol) + + @is_context_parallel @require_torch_multi_accelerator class ContextParallelTesterMixin: diff --git a/tests/models/transformers/test_models_transformer_flux.py b/tests/models/transformers/test_models_transformer_flux.py index b08fde296320..04542253c7f6 100644 --- a/tests/models/transformers/test_models_transformer_flux.py +++ b/tests/models/transformers/test_models_transformer_flux.py @@ -53,6 +53,7 @@ SingleFileTesterMixin, TaylorSeerCacheTesterMixin, TensorParallelTesterMixin, + TensorParallelTPUTesterMixin, TorchAoCompileTesterMixin, TorchAoTesterMixin, TorchCompileTesterMixin, @@ -276,6 +277,14 @@ class TestFluxTransformerTensorParallel(FluxTransformerTesterConfig, TensorParal """Tensor Parallel inference tests for Flux Transformer (CUDA/XPU multi-accelerator).""" +class TestFluxTransformerTensorParallelTPU(FluxTransformerTesterConfig, TensorParallelTPUTesterMixin): + """Tensor Parallel inference test for Flux Transformer on TPU.""" + + def get_init_dict(self): + # One head per chip. + return {**super().get_init_dict(), "num_attention_heads": self.tp_world_size} + + def make_neuron_tp_spec(): """Model spec consumed by the generic Neuron TP worker (`_neuron_tp_worker.py`). diff --git a/tests/models/transformers/test_models_transformer_flux2.py b/tests/models/transformers/test_models_transformer_flux2.py index 3263ce68202c..68e460e33c4b 100644 --- a/tests/models/transformers/test_models_transformer_flux2.py +++ b/tests/models/transformers/test_models_transformer_flux2.py @@ -42,6 +42,7 @@ ModelTesterMixin, SingleFileTesterMixin, TensorParallelTesterMixin, + TensorParallelTPUTesterMixin, TorchAoCompileTesterMixin, TorchAoTesterMixin, TorchCompileTesterMixin, @@ -176,6 +177,14 @@ def make_neuron_tp_spec(): return Flux2Transformer2DModel, config.get_init_dict(), config.get_dummy_inputs(device="cpu") +class TestFlux2TransformerTensorParallelTPU(Flux2TransformerTesterConfig, TensorParallelTPUTesterMixin): + """Tensor Parallel inference test for Flux2 Transformer on TPU.""" + + def get_init_dict(self): + # One head per chip. + return {**super().get_init_dict(), "num_attention_heads": self.tp_world_size} + + @is_tensor_parallel @require_torch_neuron class TestFlux2TransformerTensorParallelNeuron: diff --git a/tests/models/transformers/test_models_transformer_qwenimage.py b/tests/models/transformers/test_models_transformer_qwenimage.py index 5fcf37f6ff3f..6210aecff87f 100644 --- a/tests/models/transformers/test_models_transformer_qwenimage.py +++ b/tests/models/transformers/test_models_transformer_qwenimage.py @@ -37,6 +37,7 @@ MemoryTesterMixin, ModelTesterMixin, TensorParallelTesterMixin, + TensorParallelTPUTesterMixin, TorchAoTesterMixin, TorchCompileTesterMixin, TrainingTesterMixin, @@ -307,6 +308,18 @@ class TestQwenImageTransformerTensorParallel(QwenImageTransformerTesterConfig, T """Tensor Parallel inference tests for QwenImage Transformer (CUDA/XPU multi-accelerator).""" +class TestQwenImageTransformerTensorParallelTPU(QwenImageTransformerTesterConfig, TensorParallelTPUTesterMixin): + """Tensor Parallel inference test for QwenImage Transformer on TPU.""" + + def get_init_dict(self): + # One head per chip. + return {**super().get_init_dict(), "num_attention_heads": self.tp_world_size} + + def test_tensor_parallel_tpu_inference(self): + # TPU numerics differ by ~1e-2 between sharded and unsharded QwenImage (~1e-7 on CPU). + super().test_tensor_parallel_tpu_inference(atol=2e-2, rtol=2e-2) + + def make_neuron_tp_spec(): """Model spec consumed by the generic Neuron TP worker (``_neuron_tp_worker.py``). diff --git a/tests/testing_utils.py b/tests/testing_utils.py index cce6f15325b4..a22da6de08e5 100644 --- a/tests/testing_utils.py +++ b/tests/testing_utils.py @@ -46,6 +46,7 @@ is_timm_available, is_torch_available, is_torch_neuronx_available, + is_torch_tpu_available, is_torch_version, is_torchao_available, is_torchsde_available, @@ -566,6 +567,14 @@ def require_torch_neuron(test_case): )(test_case) +def require_torch_tpu(test_case): + """Decorator marking a test that requires a TPU device (torch_tpu).""" + return pytest.mark.skipif( + not is_torch_tpu_available(), + reason="test requires TPU device (torch_tpu)", + )(test_case) + + def require_torch_multi_gpu(test_case): """ Decorator marking a test that requires a multi-GPU setup (in PyTorch). These tests are skipped on a machine without