api cancellation

closing the http request to the api now - sends a cancellation from the api - writes that canellation in the master - worker plans off the cancellation - runner observes that cancellation after every generation step (+1 communication per token) - cancellation happens synchronously to prevent gpu locks
2026-02-05 19:52:16 -05:00 · 2026-02-05 18:19:32 +00:00
36 changed files with 1329 additions and 2185 deletions
--- a/.mlx_typings/mflux/models/flux/variants/kontext/init.pyi
+++ b/.mlx_typings/mflux/models/flux/variants/kontext/init.pyi
@@ -1,7 +0,0 @@
-"""
-This type stub file was generated by pyright.
-"""
-
-from mflux.models.flux.variants.kontext.flux_kontext import Flux1Kontext
-
-__all__ = ["Flux1Kontext"]
--- a/.mlx_typings/mflux/models/flux/variants/kontext/flux_kontext.pyi
+++ b/.mlx_typings/mflux/models/flux/variants/kontext/flux_kontext.pyi
@@ -1,49 +0,0 @@
-"""
-This type stub file was generated by pyright.
-"""
-
-from pathlib import Path
-from typing import Any
-
-from mlx import nn
-
-from mflux.models.common.config.model_config import ModelConfig
-from mflux.models.flux.model.flux_text_encoder.clip_encoder.clip_encoder import (
-    CLIPEncoder,
-)
-from mflux.models.flux.model.flux_text_encoder.t5_encoder.t5_encoder import T5Encoder
-from mflux.models.flux.model.flux_transformer.transformer import Transformer
-from mflux.models.flux.model.flux_vae.vae import VAE
-from mflux.utils.generated_image import GeneratedImage
-
-class Flux1Kontext(nn.Module):
-    vae: VAE
-    transformer: Transformer
-    t5_text_encoder: T5Encoder
-    clip_text_encoder: CLIPEncoder
-    bits: int | None
-    lora_paths: list[str] | None
-    lora_scales: list[float] | None
-    prompt_cache: dict[str, Any]
-    tokenizers: dict[str, Any]
-
-    def __init__(
-        self,
-        quantize: int | None = ...,
-        model_path: str | None = ...,
-        lora_paths: list[str] | None = ...,
-        lora_scales: list[float] | None = ...,
-        model_config: ModelConfig = ...,
-    ) -> None: ...
-    def generate_image(
-        self,
-        seed: int,
-        prompt: str,
-        num_inference_steps: int = ...,
-        height: int = ...,
-        width: int = ...,
-        guidance: float = ...,
-        image_path: Path | str | None = ...,
-        image_strength: float | None = ...,
-        scheduler: str = ...,
-    ) -> GeneratedImage: ...
--- a/.mlx_typings/mflux/models/flux/variants/kontext/kontext_util.pyi
+++ b/.mlx_typings/mflux/models/flux/variants/kontext/kontext_util.pyi
@@ -1,16 +0,0 @@
-"""
-This type stub file was generated by pyright.
-"""
-
-import mlx.core as mx
-
-from mflux.models.flux.model.flux_vae.vae import VAE
-
-class KontextUtil:
-    @staticmethod
-    def create_image_conditioning_latents(
-        vae: VAE,
-        height: int,
-        width: int,
-        image_path: str,
-    ) -> tuple[mx.array, mx.array]: ...
--- a/MISSED_THINGS.md
+++ b/MISSED_THINGS.md
@@ -5,21 +5,21 @@
 [X] Fetching download status of all models on start
 [X] Deduplication of tasks in plan_step.
 [X] resolve_allow_patterns should just be wildcard now.
-[] no mx_barrier in genreate.py mlx_generate at the end.
+[X] no mx_barrier in genreate.py mlx_generate at the end.
 [] cache assertion not needed in auto_parallel.py PipelineLastLayer.
-[] GPTOSS support dropped in auto_parallel.py.
-[] sharding changed "all-to-sharded" became _all_to_sharded in auto_parallel.py.
-[] same as above with "sharded-to-all" became _sharded_to_all in auto_parallel.py.
-[] Dropped support for Ministral3Model, DeepseekV32Model, Glm4MoeModel, Qwen3NextModel, GptOssMode in auto_parallel.py.
+[X] GPTOSS support dropped in auto_parallel.py.
+[X] sharding changed "all-to-sharded" became _all_to_sharded in auto_parallel.py.
+[X] same as above with "sharded-to-all" became _sharded_to_all in auto_parallel.py.
+[X] Dropped support for Ministral3Model, DeepseekV32Model, Glm4MoeModel, Qwen3NextModel, GptOssMode in auto_parallel.py.
 [] Dropped prefill/decode code in auto_parallel.py and utils_mlx.py.
 [X] KV_CACHE_BITS should be None to disable quantized KV cache.
-[] Dropped _set_nofile_limit in utils_mlx.py.
-[] We have group optional in load_mlx_items in utils_mlx.py.
-[] Dropped add_missing_chat_templates for GptOss in load_mlx_items in utils_mlx.py.
-[] Dropped model.make_cache in make_kv_cache in utils_mlx.py.
+[X] Dropped _set_nofile_limit in utils_mlx.py.
+[X] We have group optional in load_mlx_items in utils_mlx.py.
+[X] Dropped add_missing_chat_templates for GptOss in load_mlx_items in utils_mlx.py.
+[X] Dropped model.make_cache in make_kv_cache in utils_mlx.py.
 [X] We put cache limit back in utils_mlx.py.
-[] topology.py remove_node removes the connections after checking if node is is in self._node_id_to_rx_id_map. on beta_1 it checks after, so would remove stale connections I guess?
-[] Missing Glm 4.7 model cards (this isn't ready yet but should be picked up, probably create an issue... the blocker is transforemrs version doesn't support the tokenizer for Glm 4.7. rc-1 does but we can't upgrade as it breaks other things.)
+[X] topology.py remove_node removes the connections after checking if node is is in self._node_id_to_rx_id_map. on beta_1 it checks after, so would remove stale connections I guess?
+[X] Missing Glm 4.7 model cards (this isn't ready yet but should be picked up, probably create an issue... the blocker is transforemrs version doesn't support the tokenizer for Glm 4.7. rc-1 does but we can't upgrade as it breaks other things.)
 [] try-except in _command_processor only excepts ValueError. This was silently failing leading to un-debuggable errors (we had a KeyError that was happening ). Changed this to catch Exception instead of ValueError. See exo-v2 89ae38405e0052e3c22405daf094b065878aa873 and fb99fea69b5a39017efc90c5dad0072e677455f0.
 [X] In placement.py, place_instance no longer looks at model_meta.supports_tensor and check if this tensor parallel number of nodes is supported by the model's tensor dimensions.
 [X] In placement.py, place_instanec, we no longer have the special case to exclude DeepSeek v3.1 pipeline parallel (it doesn't work).
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -26,7 +26,7 @@ dependencies = [
    "httpx>=0.28.1",
    "tomlkit>=0.14.0",
    "pillow>=11.0,<12.0", # compatibility with mflux
-    "mflux==0.15.5",
+    "mflux==0.15.4",
    "python-multipart>=0.0.21",
 ]

--- a/resources/image_model_cards/exolabs--FLUX.1-Kontext-dev-4bit.toml
+++ b/resources/image_model_cards/exolabs--FLUX.1-Kontext-dev-4bit.toml
@@ -1,45 +0,0 @@
-model_id = "exolabs/FLUX.1-Kontext-dev-4bit"
-n_layers = 57
-hidden_size = 1
-supports_tensor = false
-tasks = ["ImageToImage"]
-
-[storage_size]
-in_bytes = 15475325472
-
-[[components]]
-component_name = "text_encoder"
-component_path = "text_encoder/"
-n_layers = 12
-can_shard = false
-
-[components.storage_size]
-in_bytes = 0
-
-[[components]]
-component_name = "text_encoder_2"
-component_path = "text_encoder_2/"
-n_layers = 24
-can_shard = false
-safetensors_index_filename = "model.safetensors.index.json"
-
-[components.storage_size]
-in_bytes = 9524621312
-
-[[components]]
-component_name = "transformer"
-component_path = "transformer/"
-n_layers = 57
-can_shard = true
-safetensors_index_filename = "diffusion_pytorch_model.safetensors.index.json"
-
-[components.storage_size]
-in_bytes = 5950704160
-
-[[components]]
-component_name = "vae"
-component_path = "vae/"
-can_shard = false
-
-[components.storage_size]
-in_bytes = 0
--- a/resources/image_model_cards/exolabs--FLUX.1-Kontext-dev-8bit.toml
+++ b/resources/image_model_cards/exolabs--FLUX.1-Kontext-dev-8bit.toml
@@ -1,45 +0,0 @@
-model_id = "exolabs/FLUX.1-Kontext-dev-8bit"
-n_layers = 57
-hidden_size = 1
-supports_tensor = false
-tasks = ["ImageToImage"]
-
-[storage_size]
-in_bytes = 21426029632
-
-[[components]]
-component_name = "text_encoder"
-component_path = "text_encoder/"
-n_layers = 12
-can_shard = false
-
-[components.storage_size]
-in_bytes = 0
-
-[[components]]
-component_name = "text_encoder_2"
-component_path = "text_encoder_2/"
-n_layers = 24
-can_shard = false
-safetensors_index_filename = "model.safetensors.index.json"
-
-[components.storage_size]
-in_bytes = 9524621312
-
-[[components]]
-component_name = "transformer"
-component_path = "transformer/"
-n_layers = 57
-can_shard = true
-safetensors_index_filename = "diffusion_pytorch_model.safetensors.index.json"
-
-[components.storage_size]
-in_bytes = 11901408320
-
-[[components]]
-component_name = "vae"
-component_path = "vae/"
-can_shard = false
-
-[components.storage_size]
-in_bytes = 0
--- a/resources/image_model_cards/exolabs--FLUX.1-Kontext-dev.toml
+++ b/resources/image_model_cards/exolabs--FLUX.1-Kontext-dev.toml
@@ -1,45 +0,0 @@
-model_id = "exolabs/FLUX.1-Kontext-dev"
-n_layers = 57
-hidden_size = 1
-supports_tensor = false
-tasks = ["ImageToImage"]
-
-[storage_size]
-in_bytes = 33327437952
-
-[[components]]
-component_name = "text_encoder"
-component_path = "text_encoder/"
-n_layers = 12
-can_shard = false
-
-[components.storage_size]
-in_bytes = 0
-
-[[components]]
-component_name = "text_encoder_2"
-component_path = "text_encoder_2/"
-n_layers = 24
-can_shard = false
-safetensors_index_filename = "model.safetensors.index.json"
-
-[components.storage_size]
-in_bytes = 9524621312
-
-[[components]]
-component_name = "transformer"
-component_path = "transformer/"
-n_layers = 57
-can_shard = true
-safetensors_index_filename = "diffusion_pytorch_model.safetensors.index.json"
-
-[components.storage_size]
-in_bytes = 23802816640
-
-[[components]]
-component_name = "vae"
-component_path = "vae/"
-can_shard = false
-
-[components.storage_size]
-in_bytes = 0
--- a/src/exo/master/adapters/chat_completions.py
+++ b/src/exo/master/adapters/chat_completions.py
@@ -176,7 +176,7 @@ async def generate_chat_stream(
 async def collect_chat_response(
    command_id: CommandId,
    chunk_stream: AsyncGenerator[ErrorChunk | ToolCallChunk | TokenChunk, None],
-) -> ChatCompletionResponse:
+) -> AsyncGenerator[str]:
    """Collect all token chunks and return a single ChatCompletionResponse."""
    text_parts: list[str] = []
    tool_calls: list[ToolCall] = []
@@ -223,7 +223,7 @@ async def collect_chat_response(
    combined_text = "".join(text_parts)
    assert model is not None

-    return ChatCompletionResponse(
+    yield ChatCompletionResponse(
        id=command_id,
        created=int(time.time()),
        model=model,
@@ -241,4 +241,5 @@ async def collect_chat_response(
                finish_reason=finish_reason,
            )
        ],
-    )
+    ).model_dump_json()
+    return
--- a/src/exo/master/api.py
+++ b/src/exo/master/api.py
@@ -123,6 +123,7 @@ from exo.shared.types.commands import (
    PlaceInstance,
    SendInputChunk,
    StartDownload,
+    TaskCancelled,
    TaskFinished,
    TextGeneration,
 )
@@ -529,16 +530,14 @@ class API:
                        break

        except anyio.get_cancelled_exc_class():
-            # TODO: TaskCancelled
-            """
-            self.command_sender.send_nowait(
-                ForwarderCommand(origin=self.node_id, command=command)
-            )
-            """
+            command = TaskCancelled(cancelled_command_id=command_id)
+            with anyio.CancelScope(shield=True):
+                await self.command_sender.send(
+                    ForwarderCommand(origin=self.node_id, command=command)
+                )
            raise
        finally:
-            command = TaskFinished(finished_command_id=command_id)
-            await self._send(command)
+            await self._send(TaskFinished(finished_command_id=command_id))
            if command_id in self._text_generation_queues:
                del self._text_generation_queues[command_id]

@@ -633,11 +632,14 @@ class API:
                    "X-Accel-Buffering": "no",
                },
            )
-
-        return await collect_chat_response(
-            command.command_id,
-            self._token_chunk_stream(command.command_id),
-        )
+        else:
+            return StreamingResponse(
+                collect_chat_response(
+                    command.command_id,
+                    self._token_chunk_stream(command.command_id),
+                ),
+                media_type="application/json",
+            )

    async def bench_chat_completions(
        self, payload: BenchChatCompletionRequest
@@ -653,8 +655,7 @@ class API:
        command = TextGeneration(task_params=task_params)
        await self._send(command)

-        response = await self._collect_text_generation_with_stats(command.command_id)
-        return response
+        return await self._collect_text_generation_with_stats(command.command_id)

    async def _resolve_and_validate_text_model(self, model_id: ModelId) -> ModelId:
        """Validate a text model exists and return the resolved model ID.
@@ -856,6 +857,11 @@ class API:
                        del image_metadata[key]

        except anyio.get_cancelled_exc_class():
+            command = TaskCancelled(cancelled_command_id=command_id)
+            with anyio.CancelScope(shield=True):
+                await self.command_sender.send(
+                    ForwarderCommand(origin=self.node_id, command=command)
+                )
            raise
        finally:
            await self._send(TaskFinished(finished_command_id=command_id))
@@ -937,6 +943,11 @@ class API:

            return (images, stats if capture_stats else None)
        except anyio.get_cancelled_exc_class():
+            command = TaskCancelled(cancelled_command_id=command_id)
+            with anyio.CancelScope(shield=True):
+                await self.command_sender.send(
+                    ForwarderCommand(origin=self.node_id, command=command)
+                )
            raise
        finally:
            await self._send(TaskFinished(finished_command_id=command_id))
--- a/src/exo/master/main.py
+++ b/src/exo/master/main.py
@@ -21,6 +21,7 @@ from exo.shared.types.commands import (
    PlaceInstance,
    RequestEventLog,
    SendInputChunk,
+    TaskCancelled,
    TaskFinished,
    TestCommand,
    TextGeneration,
@@ -36,6 +37,7 @@ from exo.shared.types.events import (
    NodeTimedOut,
    TaskCreated,
    TaskDeleted,
+    TaskStatusUpdated,
    TraceEventData,
    TracesCollected,
    TracesMerged,
@@ -278,7 +280,7 @@ class Master:
                        case DeleteInstance():
                            placement = delete_instance(command, self.state.instances)
                            transition_events = get_transition_events(
-                                self.state.instances, placement
+                                self.state.instances, placement, self.state.tasks
                            )
                            generated_events.extend(transition_events)
                        case PlaceInstance():
@@ -290,7 +292,7 @@ class Master:
                                self.state.node_network,
                            )
                            transition_events = get_transition_events(
-                                self.state.instances, placement
+                                self.state.instances, placement, self.state.tasks
                            )
                            generated_events.extend(transition_events)
                        case CreateInstance():
@@ -300,7 +302,7 @@ class Master:
                                self.state.instances,
                            )
                            transition_events = get_transition_events(
-                                self.state.instances, placement
+                                self.state.instances, placement, self.state.tasks
                            )
                            generated_events.extend(transition_events)
                        case SendInputChunk(chunk=chunk):
@@ -310,6 +312,18 @@ class Master:
                                    chunk=chunk,
                                )
                            )
+                        case TaskCancelled():
+                            if (
+                                task_id := self.command_task_mapping.get(
+                                    command.cancelled_command_id
+                                )
+                            ) is not None:
+                                generated_events.append(
+                                    TaskStatusUpdated(
+                                        task_status=TaskStatus.Cancelled,
+                                        task_id=task_id,
+                                    )
+                                )
                        case TaskFinished():
                            generated_events.append(
                                TaskDeleted(
@@ -318,10 +332,9 @@ class Master:
                                    ]
                                )
                            )
-                            if command.finished_command_id in self.command_task_mapping:
-                                del self.command_task_mapping[
-                                    command.finished_command_id
-                                ]
+                            self.command_task_mapping.pop(
+                                command.finished_command_id, None
+                            )
                        case RequestEventLog():
                            # We should just be able to send everything, since other buffers will ignore old messages
                            for i in range(command.since_idx, len(self._event_log)):
--- a/src/exo/master/placement.py
+++ b/src/exo/master/placement.py
@@ -20,9 +20,15 @@ from exo.shared.types.commands import (
    PlaceInstance,
 )
 from exo.shared.types.common import NodeId
-from exo.shared.types.events import Event, InstanceCreated, InstanceDeleted
+from exo.shared.types.events import (
+    Event,
+    InstanceCreated,
+    InstanceDeleted,
+    TaskStatusUpdated,
+)
 from exo.shared.types.memory import Memory
 from exo.shared.types.profiling import MemoryUsage, NodeNetworkInfo
+from exo.shared.types.tasks import Task, TaskId, TaskStatus
 from exo.shared.types.worker.instances import (
    Instance,
    InstanceId,
@@ -180,6 +186,7 @@ def delete_instance(
 def get_transition_events(
    current_instances: Mapping[InstanceId, Instance],
    target_instances: Mapping[InstanceId, Instance],
+    tasks: Mapping[TaskId, Task],
 ) -> Sequence[Event]:
    events: list[Event] = []

@@ -195,6 +202,18 @@ def get_transition_events(
    # find instances to delete
    for instance_id in current_instances:
        if instance_id not in target_instances:
+            for task in tasks.values():
+                if task.instance_id == instance_id and task.task_status in [
+                    TaskStatus.Pending,
+                    TaskStatus.Running,
+                ]:
+                    events.append(
+                        TaskStatusUpdated(
+                            task_status=TaskStatus.Cancelled,
+                            task_id=task.task_id,
+                        )
+                    )
+
            events.append(
                InstanceDeleted(
                    instance_id=instance_id,
--- a/src/exo/master/tests/test_placement.py
+++ b/src/exo/master/tests/test_placement.py
@@ -239,7 +239,7 @@ def test_get_transition_events_no_change(instance: Instance):
    target_instances = {instance_id: instance}

    # act
-    events = get_transition_events(current_instances, target_instances)
+    events = get_transition_events(current_instances, target_instances, {})

    # assert
    assert len(events) == 0
@@ -252,7 +252,7 @@ def test_get_transition_events_create_instance(instance: Instance):
    target_instances: dict[InstanceId, Instance] = {instance_id: instance}

    # act
-    events = get_transition_events(current_instances, target_instances)
+    events = get_transition_events(current_instances, target_instances, {})

    # assert
    assert len(events) == 1
@@ -266,7 +266,7 @@ def test_get_transition_events_delete_instance(instance: Instance):
    target_instances: dict[InstanceId, Instance] = {}

    # act
-    events = get_transition_events(current_instances, target_instances)
+    events = get_transition_events(current_instances, target_instances, {})

    # assert
    assert len(events) == 1
--- a/src/exo/shared/types/commands.py
+++ b/src/exo/shared/types/commands.py
@@ -48,6 +48,10 @@ class DeleteInstance(BaseCommand):
    instance_id: InstanceId


+class TaskCancelled(BaseCommand):
+    cancelled_command_id: CommandId
+
+
 class TaskFinished(BaseCommand):
    finished_command_id: CommandId

@@ -84,6 +88,7 @@ Command = (
    | PlaceInstance
    | CreateInstance
    | DeleteInstance
+    | TaskCancelled
    | TaskFinished
    | SendInputChunk
 )
--- a/src/exo/shared/types/tasks.py
+++ b/src/exo/shared/types/tasks.py
@@ -24,6 +24,7 @@ class TaskStatus(str, Enum):
    Complete = "Complete"
    TimedOut = "TimedOut"
    Failed = "Failed"
+    Cancelled = "Cancelled"


 class BaseTask(TaggedModel):
@@ -60,6 +61,11 @@ class TextGeneration(BaseTask):  # emitted by Master
    error_message: str | None = Field(default=None)


+class CancelTask(BaseTask):
+    cancelled_task_id: TaskId
+    runner_id: RunnerId
+
+
 class ImageGeneration(BaseTask):  # emitted by Master
    command_id: CommandId
    task_params: ImageGenerationTaskParams
@@ -87,6 +93,7 @@ Task = (
    | LoadModel
    | StartWarmup
    | TextGeneration
+    | CancelTask
    | ImageGeneration
    | ImageEdits
    | Shutdown
--- a/src/exo/utils/channels.py
+++ b/src/exo/utils/channels.py
@@ -125,7 +125,9 @@ class MpSender[T]:
            self._state.buffer.put(item, block=True)

    async def send_async(self, item: T) -> None:
-        await to_thread.run_sync(self.send, item, limiter=CapacityLimiter(1))
+        await to_thread.run_sync(
+            self.send, item, limiter=CapacityLimiter(1), abandon_on_cancel=True
+        )

    def close(self) -> None:
        if not self._state.closed.is_set():
--- a/src/exo/worker/engines/image/models/init.py
+++ b/src/exo/worker/engines/image/models/init.py
@@ -5,9 +5,7 @@ from exo.worker.engines.image.config import ImageModelConfig
 from exo.worker.engines.image.models.base import ModelAdapter
 from exo.worker.engines.image.models.flux import (
    FLUX_DEV_CONFIG,
-    FLUX_KONTEXT_CONFIG,
    FLUX_SCHNELL_CONFIG,
-    FluxKontextModelAdapter,
    FluxModelAdapter,
 )
 from exo.worker.engines.image.models.qwen import (
@@ -28,16 +26,13 @@ AdapterFactory = Callable[
 # Registry maps model_family string to adapter factory
 _ADAPTER_REGISTRY: dict[str, AdapterFactory] = {
    "flux": FluxModelAdapter,
-    "flux-kontext": FluxKontextModelAdapter,
    "qwen-edit": QwenEditModelAdapter,
    "qwen": QwenModelAdapter,
 }

 # Config registry: maps model ID patterns to configs
-# Order matters: longer/more-specific patterns must come before shorter ones
 _CONFIG_REGISTRY: dict[str, ImageModelConfig] = {
    "flux.1-schnell": FLUX_SCHNELL_CONFIG,
-    "flux.1-kontext": FLUX_KONTEXT_CONFIG,  # Must come before "flux.1-dev" for pattern matching
    "flux.1-krea-dev": FLUX_DEV_CONFIG,  # Must come before "flux.1-dev" for pattern matching
    "flux.1-dev": FLUX_DEV_CONFIG,
    "qwen-image-edit": QWEN_IMAGE_EDIT_CONFIG,  # Must come before "qwen-image" for pattern matching
--- a/src/exo/worker/engines/image/models/base.py
+++ b/src/exo/worker/engines/image/models/base.py
@@ -66,19 +66,6 @@ class PromptData(ABC):
        """
        ...

-    @property
-    @abstractmethod
-    def kontext_image_ids(self) -> mx.array | None:
-        """Kontext-style position IDs for image conditioning.
-
-        For FLUX.1-Kontext models, returns position IDs with first_coord=1
-        to distinguish conditioning tokens from generation tokens (first_coord=0).
-
-        Returns:
-            Position IDs array [1, seq_len, 3] for Kontext, None for other models.
-        """
-        ...
-
    @abstractmethod
    def get_batched_cfg_data(
        self,
--- a/src/exo/worker/engines/image/models/flux/init.py
+++ b/src/exo/worker/engines/image/models/flux/init.py
@@ -1,17 +1,11 @@
 from exo.worker.engines.image.models.flux.adapter import FluxModelAdapter
 from exo.worker.engines.image.models.flux.config import (
    FLUX_DEV_CONFIG,
-    FLUX_KONTEXT_CONFIG,
    FLUX_SCHNELL_CONFIG,
 )
-from exo.worker.engines.image.models.flux.kontext_adapter import (
-    FluxKontextModelAdapter,
-)

 __all__ = [
    "FluxModelAdapter",
-    "FluxKontextModelAdapter",
    "FLUX_DEV_CONFIG",
-    "FLUX_KONTEXT_CONFIG",
    "FLUX_SCHNELL_CONFIG",
 ]
--- a/src/exo/worker/engines/image/models/flux/adapter.py
+++ b/src/exo/worker/engines/image/models/flux/adapter.py
@@ -59,10 +59,6 @@ class FluxPromptData(PromptData):
    def conditioning_latents(self) -> mx.array | None:
        return None

-    @property
-    def kontext_image_ids(self) -> mx.array | None:
-        return None
-
    def get_batched_cfg_data(
        self,
    ) -> tuple[mx.array, mx.array, mx.array | None, mx.array | None] | None:
--- a/src/exo/worker/engines/image/models/flux/config.py
+++ b/src/exo/worker/engines/image/models/flux/config.py
@@ -32,19 +32,3 @@ FLUX_DEV_CONFIG = ImageModelConfig(
    default_steps={"low": 10, "medium": 25, "high": 50},
    num_sync_steps_factor=0.125,  # ~3 sync steps for medium (25 steps)
 )
-
-
-FLUX_KONTEXT_CONFIG = ImageModelConfig(
-    model_family="flux-kontext",
-    block_configs=(
-        TransformerBlockConfig(
-            block_type=BlockType.JOINT, count=19, has_separate_text_output=True
-        ),
-        TransformerBlockConfig(
-            block_type=BlockType.SINGLE, count=38, has_separate_text_output=False
-        ),
-    ),
-    default_steps={"low": 10, "medium": 25, "high": 50},
-    num_sync_steps_factor=0.125,  # ~3 sync steps for medium (25 steps)
-    guidance_scale=4.0,
-)
--- a/src/exo/worker/engines/image/models/flux/kontext_adapter.py
+++ b/src/exo/worker/engines/image/models/flux/kontext_adapter.py
@@ -1,348 +0,0 @@
-import math
-from pathlib import Path
-from typing import Any, final
-
-import mlx.core as mx
-from mflux.models.common.config.config import Config
-from mflux.models.common.config.model_config import ModelConfig
-from mflux.models.flux.latent_creator.flux_latent_creator import FluxLatentCreator
-from mflux.models.flux.model.flux_text_encoder.prompt_encoder import PromptEncoder
-from mflux.models.flux.model.flux_transformer.transformer import Transformer
-from mflux.models.flux.variants.kontext.flux_kontext import Flux1Kontext
-from mflux.models.flux.variants.kontext.kontext_util import KontextUtil
-
-from exo.worker.engines.image.config import ImageModelConfig
-from exo.worker.engines.image.models.base import (
-    ModelAdapter,
-    PromptData,
-    RotaryEmbeddings,
-)
-from exo.worker.engines.image.models.flux.wrappers import (
-    FluxJointBlockWrapper,
-    FluxSingleBlockWrapper,
-)
-from exo.worker.engines.image.pipeline.block_wrapper import (
-    JointBlockWrapper,
-    SingleBlockWrapper,
-)
-
-
-@final
-class FluxKontextPromptData(PromptData):
-    """Prompt data for FLUX.1-Kontext image editing.
-
-    Stores text embeddings along with conditioning latents and position IDs
-    for the input image.
-    """
-
-    def __init__(
-        self,
-        prompt_embeds: mx.array,
-        pooled_prompt_embeds: mx.array,
-        conditioning_latents: mx.array,
-        kontext_image_ids: mx.array,
-    ):
-        self._prompt_embeds = prompt_embeds
-        self._pooled_prompt_embeds = pooled_prompt_embeds
-        self._conditioning_latents = conditioning_latents
-        self._kontext_image_ids = kontext_image_ids
-
-    @property
-    def prompt_embeds(self) -> mx.array:
-        return self._prompt_embeds
-
-    @property
-    def pooled_prompt_embeds(self) -> mx.array:
-        return self._pooled_prompt_embeds
-
-    @property
-    def negative_prompt_embeds(self) -> mx.array | None:
-        return None
-
-    @property
-    def negative_pooled_prompt_embeds(self) -> mx.array | None:
-        return None
-
-    def get_encoder_hidden_states_mask(self, positive: bool = True) -> mx.array | None:
-        return None
-
-    @property
-    def cond_image_grid(
-        self,
-    ) -> tuple[int, int, int] | list[tuple[int, int, int]] | None:
-        return None
-
-    @property
-    def conditioning_latents(self) -> mx.array | None:
-        """VAE-encoded input image latents for Kontext conditioning."""
-        return self._conditioning_latents
-
-    @property
-    def kontext_image_ids(self) -> mx.array | None:
-        """Position IDs for Kontext conditioning (first_coord=1)."""
-        return self._kontext_image_ids
-
-    def get_cfg_branch_data(
-        self, positive: bool
-    ) -> tuple[mx.array, mx.array | None, mx.array | None, mx.array | None]:
-        """Kontext doesn't use CFG, but we return positive data for compatibility."""
-        return (
-            self._prompt_embeds,
-            None,
-            self._pooled_prompt_embeds,
-            self._conditioning_latents,
-        )
-
-    def get_batched_cfg_data(
-        self,
-    ) -> tuple[mx.array, mx.array, mx.array | None, mx.array | None] | None:
-        # Kontext doesn't use CFG
-        return None
-
-
-@final
-class FluxKontextModelAdapter(ModelAdapter[Flux1Kontext, Transformer]):
-    """Adapter for FLUX.1-Kontext image editing model.
-
-    Key differences from standard FluxModelAdapter:
-    - Takes an input image and computes output dimensions from it
-    - Creates conditioning latents from the input image via VAE
-    - Creates special position IDs (kontext_image_ids) for conditioning tokens
-    - Creates pure noise latents (not img2img blending)
-    """
-
-    def __init__(
-        self,
-        config: ImageModelConfig,
-        model_id: str,
-        local_path: Path,
-        quantize: int | None = None,
-    ):
-        self._config = config
-        self._model = Flux1Kontext(
-            model_config=ModelConfig.from_name(model_name=model_id, base_model=None),
-            model_path=str(local_path),
-            quantize=quantize,
-        )
-        self._transformer = self._model.transformer
-
-        # Stores image path and computed dimensions after set_image_dimensions
-        self._image_path: str | None = None
-        self._output_height: int | None = None
-        self._output_width: int | None = None
-
-    @property
-    def hidden_dim(self) -> int:
-        return self._transformer.x_embedder.weight.shape[0]  # pyright: ignore[reportUnknownMemberType, reportUnknownVariableType]
-
-    @property
-    def needs_cfg(self) -> bool:
-        return False
-
-    def _get_latent_creator(self) -> type:
-        return FluxLatentCreator
-
-    def get_joint_block_wrappers(
-        self,
-        text_seq_len: int,
-        encoder_hidden_states_mask: mx.array | None = None,
-    ) -> list[JointBlockWrapper[Any]]:
-        """Create wrapped joint blocks for Flux Kontext."""
-        return [
-            FluxJointBlockWrapper(block, text_seq_len)
-            for block in self._transformer.transformer_blocks
-        ]
-
-    def get_single_block_wrappers(
-        self,
-        text_seq_len: int,
-    ) -> list[SingleBlockWrapper[Any]]:
-        """Create wrapped single blocks for Flux Kontext."""
-        return [
-            FluxSingleBlockWrapper(block, text_seq_len)
-            for block in self._transformer.single_transformer_blocks
-        ]
-
-    def slice_transformer_blocks(
-        self,
-        start_layer: int,
-        end_layer: int,
-    ):
-        all_joint = list(self._transformer.transformer_blocks)
-        all_single = list(self._transformer.single_transformer_blocks)
-        total_joint_blocks = len(all_joint)
-        if end_layer <= total_joint_blocks:
-            # All assigned are joint blocks
-            joint_start, joint_end = start_layer, end_layer
-            single_start, single_end = 0, 0
-        elif start_layer >= total_joint_blocks:
-            # All assigned are single blocks
-            joint_start, joint_end = 0, 0
-            single_start = start_layer - total_joint_blocks
-            single_end = end_layer - total_joint_blocks
-        else:
-            # Spans both joint and single
-            joint_start, joint_end = start_layer, total_joint_blocks
-            single_start = 0
-            single_end = end_layer - total_joint_blocks
-
-        self._transformer.transformer_blocks = all_joint[joint_start:joint_end]
-        self._transformer.single_transformer_blocks = all_single[
-            single_start:single_end
-        ]
-
-    def set_image_dimensions(self, image_path: Path) -> tuple[int, int]:
-        """Compute and store dimensions from input image.
-
-        Also stores image_path for use in encode_prompt().
-
-        Args:
-            image_path: Path to the input image
-
-        Returns:
-            (output_width, output_height) for runtime config
-        """
-        from mflux.utils.image_util import ImageUtil
-
-        pil_image = ImageUtil.load_image(str(image_path)).convert("RGB")
-        image_size = pil_image.size
-
-        # Compute output dimensions from input image aspect ratio
-        # Target area of 1024x1024 = ~1M pixels
-        target_area = 1024 * 1024
-        ratio = image_size[0] / image_size[1]
-        output_width = math.sqrt(target_area * ratio)
-        output_height = output_width / ratio
-        output_width = round(output_width / 32) * 32
-        output_height = round(output_height / 32) * 32
-
-        # Ensure multiple of 16 for VAE
-        vae_scale_factor = 8
-        multiple_of = vae_scale_factor * 2
-        output_width = output_width // multiple_of * multiple_of
-        output_height = output_height // multiple_of * multiple_of
-
-        self._image_path = str(image_path)
-        self._output_width = int(output_width)
-        self._output_height = int(output_height)
-
-        return self._output_width, self._output_height
-
-    def create_latents(self, seed: int, runtime_config: Config) -> mx.array:
-        """Create initial noise latents for Kontext.
-
-        Unlike standard img2img which blends noise with encoded input,
-        Kontext uses pure noise latents. The input image is provided
-        separately as conditioning.
-        """
-        return FluxLatentCreator.create_noise(
-            seed=seed,
-            height=runtime_config.height,
-            width=runtime_config.width,
-        )
-
-    def encode_prompt(
-        self, prompt: str, negative_prompt: str | None = None
-    ) -> FluxKontextPromptData:
-        """Encode prompt and create conditioning from stored input image.
-
-        Must call set_image_dimensions() before this method.
-
-        Args:
-            prompt: Text prompt for editing
-            negative_prompt: Ignored (Kontext doesn't use CFG)
-
-        Returns:
-            FluxKontextPromptData with text embeddings and image conditioning
-        """
-        del negative_prompt  # Kontext doesn't support negative prompts or CFG
-
-        if (
-            self._image_path is None
-            or self._output_height is None
-            or self._output_width is None
-        ):
-            raise RuntimeError(
-                "set_image_dimensions() must be called before encode_prompt() "
-                "for FluxKontextModelAdapter"
-            )
-
-        assert isinstance(self.model.prompt_cache, dict)
-        assert isinstance(self.model.tokenizers, dict)
-
-        # Encode text prompt
-        prompt_embeds, pooled_prompt_embeds = PromptEncoder.encode_prompt(
-            prompt=prompt,
-            prompt_cache=self.model.prompt_cache,
-            t5_tokenizer=self.model.tokenizers["t5"],  # pyright: ignore[reportAny]
-            clip_tokenizer=self.model.tokenizers["clip"],  # pyright: ignore[reportAny]
-            t5_text_encoder=self.model.t5_text_encoder,
-            clip_text_encoder=self.model.clip_text_encoder,
-        )
-
-        # Create conditioning latents from input image
-        conditioning_latents, kontext_image_ids = (
-            KontextUtil.create_image_conditioning_latents(
-                vae=self.model.vae,
-                height=self._output_height,
-                width=self._output_width,
-                image_path=self._image_path,
-            )
-        )
-
-        return FluxKontextPromptData(
-            prompt_embeds=prompt_embeds,
-            pooled_prompt_embeds=pooled_prompt_embeds,
-            conditioning_latents=conditioning_latents,
-            kontext_image_ids=kontext_image_ids,
-        )
-
-    def compute_embeddings(
-        self,
-        hidden_states: mx.array,
-        prompt_embeds: mx.array,
-    ) -> tuple[mx.array, mx.array]:
-        embedded_hidden = self._transformer.x_embedder(hidden_states)
-        embedded_encoder = self._transformer.context_embedder(prompt_embeds)
-        return embedded_hidden, embedded_encoder
-
-    def compute_text_embeddings(
-        self,
-        t: int,
-        runtime_config: Config,
-        pooled_prompt_embeds: mx.array | None = None,
-        hidden_states: mx.array | None = None,
-    ) -> mx.array:
-        if pooled_prompt_embeds is None:
-            raise ValueError(
-                "pooled_prompt_embeds is required for Flux Kontext text embeddings"
-            )
-
-        return Transformer.compute_text_embeddings(
-            t, pooled_prompt_embeds, self._transformer.time_text_embed, runtime_config
-        )
-
-    def compute_rotary_embeddings(
-        self,
-        prompt_embeds: mx.array,
-        runtime_config: Config,
-        encoder_hidden_states_mask: mx.array | None = None,
-        cond_image_grid: tuple[int, int, int]
-        | list[tuple[int, int, int]]
-        | None = None,
-        kontext_image_ids: mx.array | None = None,
-    ) -> RotaryEmbeddings:
-        return Transformer.compute_rotary_embeddings(
-            prompt_embeds,
-            self._transformer.pos_embed,
-            runtime_config,
-            kontext_image_ids,
-        )
-
-    def apply_guidance(
-        self,
-        noise_positive: mx.array,
-        noise_negative: mx.array,
-        guidance_scale: float,
-    ) -> mx.array:
-        raise NotImplementedError("Flux Kontext does not use classifier-free guidance")
--- a/src/exo/worker/engines/image/models/qwen/adapter.py
+++ b/src/exo/worker/engines/image/models/qwen/adapter.py
@@ -69,10 +69,6 @@ class QwenPromptData(PromptData):
    def conditioning_latents(self) -> mx.array | None:
        return None

-    @property
-    def kontext_image_ids(self) -> mx.array | None:
-        return None
-
    def get_batched_cfg_data(
        self,
    ) -> tuple[mx.array, mx.array, mx.array | None, mx.array | None] | None:
--- a/src/exo/worker/engines/image/models/qwen/edit_adapter.py
+++ b/src/exo/worker/engines/image/models/qwen/edit_adapter.py
@@ -85,10 +85,6 @@ class QwenEditPromptData(PromptData):
    def qwen_image_ids(self) -> mx.array:
        return self._qwen_image_ids

-    @property
-    def kontext_image_ids(self) -> mx.array | None:
-        return None
-
    @property
    def is_edit_mode(self) -> bool:
        return True
--- a/src/exo/worker/engines/image/pipeline/runner.py
+++ b/src/exo/worker/engines/image/pipeline/runner.py
@@ -567,7 +567,6 @@ class DiffusionRunner:
        | list[tuple[int, int, int]]
        | None = None,
        conditioning_latents: mx.array | None = None,
-        kontext_image_ids: mx.array | None = None,
    ) -> mx.array:
        """Run a single forward pass through the transformer.
        Args:
@@ -579,7 +578,6 @@ class DiffusionRunner:
            encoder_hidden_states_mask: Attention mask for text (Qwen)
            cond_image_grid: Conditioning image grid dimensions (Qwen edit)
            conditioning_latents: Conditioning latents for edit mode
-            kontext_image_ids: Position IDs for Kontext conditioning (Flux Kontext)

        Returns:
            Noise prediction tensor
@@ -612,7 +610,6 @@ class DiffusionRunner:
            config,
            encoder_hidden_states_mask=encoder_hidden_states_mask,
            cond_image_grid=cond_image_grid,
-            kontext_image_ids=kontext_image_ids,
        )

        assert self.joint_block_wrappers is not None
@@ -684,7 +681,6 @@ class DiffusionRunner:
        prompt_data: PromptData,
    ) -> mx.array:
        cond_image_grid = prompt_data.cond_image_grid
-        kontext_image_ids = prompt_data.kontext_image_ids
        results: list[tuple[bool, mx.array]] = []

        for branch in self._get_cfg_branches(prompt_data):
@@ -704,7 +700,6 @@ class DiffusionRunner:
                encoder_hidden_states_mask=branch.mask,
                cond_image_grid=cond_image_grid,
                conditioning_latents=branch.cond_latents,
-                kontext_image_ids=kontext_image_ids,
            )
            results.append((branch.positive, noise))

@@ -907,10 +902,10 @@ class DiffusionRunner:
        config: Config,
        hidden_states: mx.array,
        prompt_data: PromptData,
+        kontext_image_ids: mx.array | None = None,
    ) -> mx.array:
        prev_latents = hidden_states
        cond_image_grid = prompt_data.cond_image_grid
-        kontext_image_ids = prompt_data.kontext_image_ids

        scaled_hidden_states = config.scheduler.scale_model_input(hidden_states, t)  # pyright: ignore[reportAny]
        original_latent_tokens: int = scaled_hidden_states.shape[1]  # pyright: ignore[reportAny]
@@ -984,10 +979,10 @@ class DiffusionRunner:
        latents: mx.array,
        prompt_data: PromptData,
        is_first_async_step: bool,
+        kontext_image_ids: mx.array | None = None,
    ) -> mx.array:
        patch_latents, token_indices = self._create_patches(latents, config)
        cond_image_grid = prompt_data.cond_image_grid
-        kontext_image_ids = prompt_data.kontext_image_ids

        prev_patch_latents = [p for p in patch_latents]

--- a/src/exo/worker/engines/mlx/generator/generate.py
+++ b/src/exo/worker/engines/mlx/generator/generate.py
@@ -32,7 +32,6 @@ from exo.worker.engines.mlx.constants import (
 )
 from exo.worker.engines.mlx.utils_mlx import (
    apply_chat_template,
-    mx_barrier,
 )
 from exo.worker.runner.bootstrap import logger

@@ -136,10 +135,6 @@ def warmup_inference(

    logger.info("Generated ALL warmup tokens")

-    # TODO: Do we want an mx_barrier?
-    #  At least this version is actively incorrect, as it should use mx_barrier(group)
-    mx_barrier()
-
    return tokens_generated


@@ -403,5 +398,3 @@ def mlx_generate(
        # Limit accumulated_text to what's needed for stop sequence detection
        if max_stop_len > 0 and len(accumulated_text) > max_stop_len:
            accumulated_text = accumulated_text[-max_stop_len:]
-
-        # TODO: Do we want an mx_barrier?
--- a/src/exo/worker/engines/mlx/utils_mlx.py
+++ b/src/exo/worker/engines/mlx/utils_mlx.py
@@ -67,8 +67,6 @@ Group = mx.distributed.Group
 resource.setrlimit(resource.RLIMIT_NOFILE, (2048, 4096))


-# TODO: Test this
-#  ALSO https://github.com/exo-explore/exo/pull/233#discussion_r2549683673
 def get_weights_size(model_shard_meta: ShardMetadata) -> Memory:
    return Memory.from_float_kb(
        (model_shard_meta.end_layer - model_shard_meta.start_layer)
@@ -86,30 +84,6 @@ class ModelLoadingTimeoutError(Exception):
    pass


-def mx_barrier(group: Group | None = None):
-    mx.eval(
-        mx.distributed.all_sum(
-            mx.array(1.0),
-            stream=mx.default_stream(mx.Device(mx.cpu)),
-            group=group,
-        )
-    )
-
-
-def broadcast_from_zero(value: int, group: Group | None = None):
-    if group is None:
-        return value
-
-    if group.rank() == 0:
-        a = mx.array([value], dtype=mx.int32)
-    else:
-        a = mx.array([0], dtype=mx.int32)
-
-    m = mx.distributed.all_sum(a, stream=mx.Device(mx.DeviceType.cpu), group=group)
-    mx.eval(m)
-    return int(m.item())
-
-
 class HostList(RootModel[list[str]]):
    @classmethod
    def from_hosts(cls, hosts: list[Host]) -> "HostList":
@@ -562,3 +536,23 @@ def mlx_cleanup(
    import gc

    gc.collect()
+
+
+def mx_any(bool_: bool, group: Group | None) -> bool:
+    if group is None:
+        return bool_
+    num_true = mx.distributed.all_sum(
+        mx.array(bool_), group=group, stream=mx.default_stream(mx.Device(mx.cpu))
+    )
+    mx.eval(num_true)
+    return num_true.item() > 0
+
+
+def mx_barrier(group: Group | None):
+    if group is None:
+        return
+    mx.eval(
+        mx.distributed.all_sum(
+            mx.array(1.0), group=group, stream=mx.default_stream(mx.Device(mx.cpu))
+        )
+    )
--- a/src/exo/worker/main.py
+++ b/src/exo/worker/main.py
@@ -32,6 +32,7 @@ from exo.shared.types.events import (
 from exo.shared.types.multiaddr import Multiaddr
 from exo.shared.types.state import State
 from exo.shared.types.tasks import (
+    CancelTask,
    CreateRunner,
    DownloadModel,
    ImageEdits,
@@ -218,15 +219,22 @@ class Worker:
                        )
                    )
                case Shutdown(runner_id=runner_id):
+                    runner = self.runners.pop(runner_id)
                    try:
                        with fail_after(3):
-                            await self.runners.pop(runner_id).start_task(task)
+                            await runner.start_task(task)
                    except TimeoutError:
                        await self.event_sender.send(
                            TaskStatusUpdated(
                                task_id=task.task_id, task_status=TaskStatus.TimedOut
                            )
                        )
+                    finally:
+                        runner.shutdown()
+                case CancelTask(
+                    cancelled_task_id=cancelled_task_id, runner_id=runner_id
+                ):
+                    await self.runners[runner_id].cancel_task(cancelled_task_id)
                case ImageEdits() if task.task_params.total_input_chunks > 0:
                    # Assemble image from chunks and inject into task
                    cmd_id = task.command_id
@@ -264,18 +272,18 @@ class Worker:
                        del self.input_chunk_buffer[cmd_id]
                    if cmd_id in self.input_chunk_counts:
                        del self.input_chunk_counts[cmd_id]
-                    await self.runners[self._task_to_runner_id(task)].start_task(
-                        modified_task
-                    )
+                    await self._start_runner_task(modified_task)
                case task:
-                    await self.runners[self._task_to_runner_id(task)].start_task(task)
+                    await self._start_runner_task(task)

    def shutdown(self):
        self._tg.cancel_scope.cancel()

-    def _task_to_runner_id(self, task: Task):
-        instance = self.state.instances[task.instance_id]
-        return instance.shard_assignments.node_to_runner[self.node_id]
+    async def _start_runner_task(self, task: Task):
+        if (instance := self.state.instances.get(task.instance_id)) is not None:
+            await self.runners[
+                instance.shard_assignments.node_to_runner[self.node_id]
+            ].start_task(task)

    async def _nack_request(self, since_idx: int) -> None:
        # We request all events after (and including) the missing index.
@@ -314,8 +322,6 @@ class Worker:
            for event in self.out_for_delivery.copy().values():
                await self.local_event_sender.send(event)

-    ## Op Executors
-
    def _create_supervisor(self, task: CreateRunner) -> RunnerSupervisor:
        """Creates and stores a new AssignedRunner with initial downloading status."""
        runner = RunnerSupervisor.create(
--- a/src/exo/worker/plan.py
+++ b/src/exo/worker/plan.py
@@ -4,6 +4,7 @@ from collections.abc import Mapping, Sequence

 from exo.shared.types.common import CommandId, NodeId
 from exo.shared.types.tasks import (
+    CancelTask,
    ConnectToGroup,
    CreateRunner,
    DownloadModel,
@@ -53,13 +54,14 @@ def plan(
 ) -> Task | None:
    # Python short circuiting OR logic should evaluate these sequentially.
    return (
-        _kill_runner(runners, all_runners, instances)
+        _cancel_tasks(runners, tasks)
+        or _kill_runner(runners, all_runners, instances)
        or _create_runner(node_id, runners, instances)
        or _model_needs_download(node_id, runners, global_download_status)
        or _init_distributed_backend(runners, all_runners)
        or _load_model(runners, all_runners, global_download_status)
        or _ready_to_warmup(runners, all_runners)
-        or _pending_tasks(runners, tasks, all_runners, input_chunk_buffer)
+        or _pending_tasks(runners, tasks, all_runners, input_chunk_buffer or {})
    )


@@ -270,7 +272,7 @@ def _pending_tasks(
    runners: Mapping[RunnerId, RunnerSupervisor],
    tasks: Mapping[TaskId, Task],
    all_runners: Mapping[RunnerId, RunnerStatus],
-    input_chunk_buffer: Mapping[CommandId, dict[int, str]] | None = None,
+    input_chunk_buffer: Mapping[CommandId, dict[int, str]],
 ) -> Task | None:
    for task in tasks.values():
        # for now, just forward chat completions
@@ -284,7 +286,7 @@ def _pending_tasks(
        if isinstance(task, ImageEdits) and task.task_params.total_input_chunks > 0:
            cmd_id = task.command_id
            expected = task.task_params.total_input_chunks
-            received = len((input_chunk_buffer or {}).get(cmd_id, {}))
+            received = len(input_chunk_buffer.get(cmd_id, {}))
            if received < expected:
                continue  # Wait for all chunks to arrive

@@ -292,16 +294,33 @@ def _pending_tasks(
            if task.instance_id != runner.bound_instance.instance.instance_id:
                continue

-            # I have a design point here; this is a state race in disguise as the task status doesn't get updated to completed fast enough
-            # however, realistically the task status should be set to completed by the LAST runner, so this is a true race
-            # the actual solution is somewhat deeper than this bypass - TODO!
+            # the task status _should_ be set to completed by the LAST runner
+            # it is currently set by the first
+            # this is definitely a hack
            if task.task_id in runner.completed:
                continue

-            # TODO: Check ordering aligns with MLX distributeds expectations.
-
            if isinstance(runner.status, RunnerReady) and all(
                isinstance(all_runners[global_runner_id], (RunnerReady, RunnerRunning))
                for global_runner_id in runner.bound_instance.instance.shard_assignments.runner_to_shard
            ):
                return task
+
+
+def _cancel_tasks(
+    runners: Mapping[RunnerId, RunnerSupervisor],
+    tasks: Mapping[TaskId, Task],
+) -> Task | None:
+    for task in tasks.values():
+        if task.task_status != TaskStatus.Cancelled:
+            continue
+        for runner_id, runner in runners.items():
+            if task.instance_id != runner.bound_instance.instance.instance_id:
+                continue
+            if task.task_id in runner.cancelled:
+                continue
+            return CancelTask(
+                instance_id=task.instance_id,
+                cancelled_task_id=task.task_id,
+                runner_id=runner_id,
+            )
--- a/src/exo/worker/runner/bootstrap.py
+++ b/src/exo/worker/runner/bootstrap.py
@@ -3,7 +3,7 @@ import os
 import loguru

 from exo.shared.types.events import Event, RunnerStatusUpdated
-from exo.shared.types.tasks import Task
+from exo.shared.types.tasks import Task, TaskId
 from exo.shared.types.worker.instances import BoundInstance, MlxJacclInstance
 from exo.shared.types.worker.runners import RunnerFailed
 from exo.utils.channels import ClosedResourceError, MpReceiver, MpSender
@@ -15,6 +15,7 @@ def entrypoint(
    bound_instance: BoundInstance,
    event_sender: MpSender[Event],
    task_receiver: MpReceiver[Task],
+    cancel_receiver: MpReceiver[TaskId],
    _logger: "loguru.Logger",
 ) -> None:
    fast_synch_override = os.environ.get("EXO_FAST_SYNCH")
@@ -38,7 +39,7 @@ def entrypoint(
    try:
        from exo.worker.runner.runner import main

-        main(bound_instance, event_sender, task_receiver)
+        main(bound_instance, event_sender, task_receiver, cancel_receiver)
    except ClosedResourceError:
        logger.warning("Runner communication closed unexpectedly")
    except Exception as e:
--- a/src/exo/worker/runner/runner.py
+++ b/src/exo/worker/runner/runner.py
@@ -1,5 +1,6 @@
 import base64
 import json
+import math
 import time
 from collections.abc import Generator
 from functools import cache
@@ -87,6 +88,7 @@ from exo.worker.engines.mlx.utils_mlx import (
    initialize_mlx,
    load_mlx_items,
    mlx_force_oom,
+    mx_any,
 )
 from exo.worker.runner.bootstrap import logger

@@ -111,6 +113,7 @@ def main(
    bound_instance: BoundInstance,
    event_sender: MpSender[Event],
    task_receiver: MpReceiver[Task],
+    cancel_receiver: MpReceiver[TaskId],
 ):
    instance, runner_id, shard_metadata = (
        bound_instance.instance,
@@ -125,11 +128,15 @@ def main(
        time.sleep(timeout)

    setup_start_time = time.time()
+    cancelled_tasks = set[TaskId]()

-    model: Model | DistributedImageModel | None = None
+    # type checker was unhappy with me - splitting these fixed it
+    inference_model: Model | None = None
+    image_model: DistributedImageModel | None = None
    tokenizer = None
    group = None
    kv_prefix_cache: KVPrefixCache | None = None
+    check_for_cancel_every: int | None = None

    current_status: RunnerStatus = RunnerIdle()
    logger.info("runner created")
@@ -142,6 +149,7 @@ def main(
            if task.task_id in seen:
                logger.warning("repeat task - potential error")
            seen.add(task.task_id)
+            cancelled_tasks.discard(TaskId("CANCEL_CURRENT_TASK"))
            event_sender.send(
                TaskStatusUpdated(task_id=task.task_id, task_status=TaskStatus.Running)
            )
@@ -187,7 +195,7 @@ def main(
                        time.sleep(0.5)

                    if ModelTask.TextGeneration in shard_metadata.model_card.tasks:
-                        model, tokenizer = load_mlx_items(
+                        inference_model, tokenizer = load_mlx_items(
                            bound_instance, group, on_timeout=on_model_load_timeout
                        )
                        logger.info(
@@ -199,7 +207,7 @@ def main(
                        ModelTask.TextToImage in shard_metadata.model_card.tasks
                        or ModelTask.ImageToImage in shard_metadata.model_card.tasks
                    ):
-                        model = initialize_image_model(bound_instance)
+                        image_model = initialize_image_model(bound_instance)
                    else:
                        raise ValueError(
                            f"Unknown model task(s): {shard_metadata.model_card.tasks}"
@@ -207,8 +215,6 @@ def main(
                    current_status = RunnerLoaded()
                    logger.info("runner loaded")
                case StartWarmup() if isinstance(current_status, RunnerLoaded):
-                    assert model
-
                    current_status = RunnerWarmingUp()
                    logger.info("runner warming up")
                    event_sender.send(
@@ -220,15 +226,30 @@ def main(

                    logger.info(f"warming up inference for instance: {instance}")
                    if ModelTask.TextGeneration in shard_metadata.model_card.tasks:
-                        assert not isinstance(model, DistributedImageModel)
+                        assert inference_model
                        assert tokenizer

+                        t = time.perf_counter()
                        toks = warmup_inference(
-                            model=model,
+                            model=inference_model,
                            tokenizer=tokenizer,
-                            # kv_prefix_cache=kv_prefix_cache,  # supply for warmup-time prefix caching
                        )
                        logger.info(f"warmed up by generating {toks} tokens")
+                        check_for_cancel_every = min(
+                            math.ceil(toks / (time.perf_counter() - t)), 100
+                        )
+                        if group is not None:
+                            check_for_cancel_every = int(
+                                mx.max(
+                                    mx.distributed.all_gather(
+                                        mx.array([check_for_cancel_every]), group=group
+                                    )
+                                ).item()
+                            )
+
+                        logger.info(
+                            f"runner checking for cancellation every {check_for_cancel_every} tokens"
+                        )
                        logger.info(
                            f"runner initialized in {time.time() - setup_start_time} seconds"
                        )
@@ -236,8 +257,8 @@ def main(
                        ModelTask.TextToImage in shard_metadata.model_card.tasks
                        or ModelTask.ImageToImage in shard_metadata.model_card.tasks
                    ):
-                        assert isinstance(model, DistributedImageModel)
-                        image = warmup_image_generator(model=model)
+                        assert image_model
+                        image = warmup_image_generator(model=image_model)
                        if image is not None:
                            logger.info(f"warmed up by generating {image.size} image")
                        else:
@@ -257,9 +278,9 @@ def main(
                        )
                    )
                    event_sender.send(TaskAcknowledged(task_id=task.task_id))
-
-                    assert model and not isinstance(model, DistributedImageModel)
+                    assert inference_model
                    assert tokenizer
+                    assert check_for_cancel_every

                    try:
                        _check_for_debug_prompts(task_params)
@@ -269,7 +290,7 @@ def main(

                        # Generate responses using the actual MLX generation
                        mlx_generator = mlx_generate(
-                            model=model,
+                            model=inference_model,
                            tokenizer=tokenizer,
                            task=task_params,
                            prompt=prompt,
@@ -293,11 +314,11 @@ def main(
                            patch_glm_tokenizer(tokenizer)

                        # GPT-OSS specific parsing to match other model formats.
-                        elif isinstance(model, GptOssModel):
+                        elif isinstance(inference_model, GptOssModel):
                            mlx_generator = parse_gpt_oss(mlx_generator)

                        if tokenizer.has_tool_calling and not isinstance(
-                            model, GptOssModel
+                            inference_model, GptOssModel
                        ):
                            assert tokenizer.tool_call_start
                            assert tokenizer.tool_call_end
@@ -310,7 +331,18 @@ def main(
                            )

                        completion_tokens = 0
+                        tokens_since_last_cancel_check = 0
                        for response in mlx_generator:
+                            tokens_since_last_cancel_check += 1
+                            if tokens_since_last_cancel_check >= check_for_cancel_every:
+                                tokens_since_last_cancel_check = 0
+                                cancelled_tasks.update(cancel_receiver.collect())
+                                want_to_cancel = (task.task_id in cancelled_tasks) or (
+                                    TaskId("CANCEL_CURRENT_TASK") in cancelled_tasks
+                                )
+                                if mx_any(want_to_cancel, group):
+                                    break
+
                            match response:
                                case GenerationResponse():
                                    completion_tokens += 1
@@ -382,7 +414,7 @@ def main(
                case ImageGeneration(
                    task_params=task_params, command_id=command_id
                ) if isinstance(current_status, RunnerReady):
-                    assert isinstance(model, DistributedImageModel)
+                    assert image_model
                    logger.info(f"received image generation request: {str(task)[:500]}")
                    current_status = RunnerRunning()
                    logger.info("runner running")
@@ -395,7 +427,9 @@ def main(

                    try:
                        image_index = 0
-                        for response in generate_image(model=model, task=task_params):
+                        for response in generate_image(
+                            model=image_model, task=task_params
+                        ):
                            is_primary_output = _is_primary_output_node(shard_metadata)

                            if is_primary_output:
@@ -445,7 +479,7 @@ def main(
                case ImageEdits(task_params=task_params, command_id=command_id) if (
                    isinstance(current_status, RunnerReady)
                ):
-                    assert isinstance(model, DistributedImageModel)
+                    assert image_model
                    logger.info(f"received image edits request: {str(task)[:500]}")
                    current_status = RunnerRunning()
                    logger.info("runner running")
@@ -458,7 +492,9 @@ def main(

                    try:
                        image_index = 0
-                        for response in generate_image(model=model, task=task_params):
+                        for response in generate_image(
+                            model=image_model, task=task_params
+                        ):
                            if _is_primary_output_node(shard_metadata):
                                match response:
                                    case PartialImageResponse():
@@ -524,7 +560,7 @@ def main(
                RunnerStatusUpdated(runner_id=runner_id, runner_status=current_status)
            )
            if isinstance(current_status, RunnerShutdown):
-                del model, tokenizer, group
+                del inference_model, image_model, tokenizer, group
                mx.clear_cache()
                import gc

--- a/src/exo/worker/runner/runner_supervisor.py
+++ b/src/exo/worker/runner/runner_supervisor.py
@@ -47,9 +47,11 @@ class RunnerSupervisor:
    _ev_recv: MpReceiver[Event]
    _task_sender: MpSender[Task]
    _event_sender: Sender[Event]
+    _cancel_sender: MpSender[TaskId]
    status: RunnerStatus = field(default_factory=RunnerIdle, init=False)
    pending: dict[TaskId, anyio.Event] = field(default_factory=dict, init=False)
    completed: set[TaskId] = field(default_factory=set, init=False)
+    cancelled: set[TaskId] = field(default_factory=set, init=False)

    @classmethod
    def create(
@@ -60,8 +62,8 @@ class RunnerSupervisor:
        initialize_timeout: float = 400,
    ) -> Self:
        ev_send, ev_recv = mp_channel[Event]()
-        # A task is kind of a runner command
        task_sender, task_recv = mp_channel[Task]()
+        cancel_sender, cancel_recv = mp_channel[TaskId]()

        runner_process = Process(
            target=entrypoint,
@@ -69,6 +71,7 @@ class RunnerSupervisor:
                bound_instance,
                ev_send,
                task_recv,
+                cancel_recv,
                logger,
            ),
            daemon=True,
@@ -83,6 +86,7 @@ class RunnerSupervisor:
            initialize_timeout=initialize_timeout,
            _ev_recv=ev_recv,
            _task_sender=task_sender,
+            _cancel_sender=cancel_sender,
            _event_sender=event_sender,
        )

@@ -97,6 +101,8 @@ class RunnerSupervisor:
        self._ev_recv.close()
        self._task_sender.close()
        self._event_sender.close()
+        self._cancel_sender.send(TaskId("CANCEL_CURRENT_TASK"))
+        self._cancel_sender.close()
        self.runner_process.join(1)
        if not self.runner_process.is_alive():
            logger.info("Runner process succesfully terminated")
@@ -112,14 +118,6 @@ class RunnerSupervisor:
        logger.critical("Runner process didn't respond to SIGTERM, killing")
        self.runner_process.kill()

-        self.runner_process.join(1)
-        if not self.runner_process.is_alive():
-            return
-
-        logger.critical(
-            "Runner process didn't respond to SIGKILL. System resources may have leaked"
-        )
-
    async def start_task(self, task: Task):
        if task.task_id in self.pending:
            logger.warning(
@@ -141,6 +139,17 @@ class RunnerSupervisor:
            return
        await event.wait()

+    async def cancel_task(self, task_id: TaskId):
+        if task_id in self.completed:
+            logger.info(f"Unable to cancel {task_id} as it has been completed")
+            return
+        self.cancelled.add(task_id)
+        with anyio.move_on_after(0.5) as scope:
+            await self._cancel_sender.send_async(task_id)
+        if scope.cancel_called:
+            logger.error("RunnerSupervisor cancel pipe blocked")
+            await self._check_runner(TimeoutError("cancel pipe blocked"))
+
    async def _forward_events(self):
        with self._ev_recv as events:
            try:
--- a/src/exo/worker/tests/unittests/test_runner/test_event_ordering.py
+++ b/src/exo/worker/tests/unittests/test_runner/test_event_ordering.py
@@ -1,7 +1,9 @@
 # Check tasks are complete before runner is ever ready.
+import unittest.mock
 from collections.abc import Iterable
 from typing import Callable

+import mlx.core as mx
 import pytest

 import exo.worker.runner.runner as mlx_runner
@@ -19,6 +21,7 @@ from exo.shared.types.tasks import (
    Shutdown,
    StartWarmup,
    Task,
+    TaskId,
    TaskStatus,
    TextGeneration,
 )
@@ -113,6 +116,7 @@ def patch_out_mlx(monkeypatch: pytest.MonkeyPatch):
    monkeypatch.setattr(mlx_runner, "load_mlx_items", make_nothin((1, MockTokenizer)))
    monkeypatch.setattr(mlx_runner, "warmup_inference", make_nothin(1))
    monkeypatch.setattr(mlx_runner, "_check_for_debug_prompts", nothin)
+    monkeypatch.setattr(mlx_runner, "mx_any", make_nothin(False))
    # Mock apply_chat_template since we're using a fake tokenizer (integer 1).
    # Returns a prompt without thinking tag so detect_thinking_prompt_suffix returns None.
    monkeypatch.setattr(mlx_runner, "apply_chat_template", make_nothin("test prompt"))
@@ -163,6 +167,7 @@ def _run(tasks: Iterable[Task]):
    )

    task_sender, task_receiver = mp_channel[Task]()
+    _cancel_sender, cancel_receiver = mp_channel[TaskId]()
    event_sender = EventCollector()

    with task_sender:
@@ -173,8 +178,11 @@ def _run(tasks: Iterable[Task]):
        # this is some c++ nonsense
        task_receiver.close = nothin
        task_receiver.join = nothin
+        with unittest.mock.patch(
+            "exo.worker.runner.runner.mx.distributed.all_gather", make_nothin(mx.array([1]))
+        ):

-        mlx_runner.main(bound_instance, event_sender, task_receiver)  # type: ignore[arg-type]
+            mlx_runner.main(bound_instance, event_sender, task_receiver, cancel_receiver)  # pyright: ignore[reportArgumentType]

        return event_sender.events

--- a/tests/run_exo_on.sh
+++ b/tests/run_exo_on.sh
@@ -35,7 +35,7 @@ i=0
 for host; do
  colour=${colours[i++ % 4]}
  ssh -T -o BatchMode=yes -o ServerAliveInterval=30 "$host@$host" \
-    "/nix/var/nix/profiles/default/bin/nix run github:exo-explore/exo/$commit" |&
+    "EXO_LIBP2P_NAMESPACE=$commit /nix/var/nix/profiles/default/bin/nix run github:exo-explore/exo/$commit" |&
    awk -v p="${colour}[${host}]${reset}" '{ print p $0; fflush() }' &
 done

--- a/tmp/quantize_and_upload.py
+++ b/tmp/quantize_and_upload.py
@@ -1,377 +0,0 @@
-#!/usr/bin/env python3
-"""
-Download an mflux model, quantize it, and upload to HuggingFace.
-
-Usage (run from mflux project directory):
-    cd /path/to/mflux
-    uv run python /path/to/quantize_and_upload.py --model black-forest-labs/FLUX.1-Kontext-dev
-    uv run python /path/to/quantize_and_upload.py --model black-forest-labs/FLUX.1-Kontext-dev --skip-base --skip-8bit
-    uv run python /path/to/quantize_and_upload.py --model black-forest-labs/FLUX.1-Kontext-dev --dry-run
-
-Requires:
-    - Must be run from mflux project directory using `uv run`
-    - huggingface_hub installed (add to mflux deps or install separately)
-    - HuggingFace authentication: run `huggingface-cli login` or set HF_TOKEN
-"""
-
-from __future__ import annotations
-
-import argparse
-import re
-import shutil
-import sys
-from pathlib import Path
-from typing import TYPE_CHECKING
-
-if TYPE_CHECKING:
-    from mflux.models.flux.variants.txt2img.flux import Flux1
-
-
-HF_ORG = "exolabs"
-
-
-def get_model_class(model_name: str) -> type:
-    """Get the appropriate model class based on model name."""
-    from mflux.models.fibo.variants.txt2img.fibo import FIBO
-    from mflux.models.flux.variants.txt2img.flux import Flux1
-    from mflux.models.flux2.variants.txt2img.flux2_klein import Flux2Klein
-    from mflux.models.qwen.variants.txt2img.qwen_image import QwenImage
-    from mflux.models.z_image.variants.turbo.z_image_turbo import ZImageTurbo
-
-    model_name_lower = model_name.lower()
-    if "qwen" in model_name_lower:
-        return QwenImage
-    elif "fibo" in model_name_lower:
-        return FIBO
-    elif "z-image" in model_name_lower or "zimage" in model_name_lower:
-        return ZImageTurbo
-    elif "flux2" in model_name_lower or "flux.2" in model_name_lower:
-        return Flux2Klein
-    else:
-        return Flux1
-
-
-def get_repo_name(model_name: str, bits: int | None) -> str:
-    """Get the HuggingFace repo name for a model variant."""
-    # Extract repo name from HF path (e.g., "black-forest-labs/FLUX.1-Kontext-dev" -> "FLUX.1-Kontext-dev")
-    base_name = model_name.split("/")[-1] if "/" in model_name else model_name
-    suffix = f"-{bits}bit" if bits else ""
-    return f"{HF_ORG}/{base_name}{suffix}"
-
-
-def get_local_path(output_dir: Path, model_name: str, bits: int | None) -> Path:
-    """Get the local save path for a model variant."""
-    # Extract repo name from HF path (e.g., "black-forest-labs/FLUX.1-Kontext-dev" -> "FLUX.1-Kontext-dev")
-    base_name = model_name.split("/")[-1] if "/" in model_name else model_name
-    suffix = f"-{bits}bit" if bits else ""
-    return output_dir / f"{base_name}{suffix}"
-
-
-def copy_source_repo(
-    source_repo: str,
-    local_path: Path,
-    dry_run: bool = False,
-) -> None:
-    """Copy all files from source repo (replicating original HF structure)."""
-    print(f"\n{'=' * 60}")
-    print(f"Copying full repo from source: {source_repo}")
-    print(f"Output path: {local_path}")
-    print(f"{'=' * 60}")
-
-    if dry_run:
-        print("[DRY RUN] Would download all files from source repo")
-        return
-
-    from huggingface_hub import snapshot_download
-
-    # Download all files to our local path
-    snapshot_download(
-        repo_id=source_repo,
-        local_dir=local_path,
-    )
-
-    # Remove root-level safetensors files (flux.1-dev.safetensors, etc.)
-    # These are redundant with the component directories
-    for f in local_path.glob("*.safetensors"):
-        print(f"Removing root-level safetensors: {f.name}")
-        if not dry_run:
-            f.unlink()
-
-    print(f"Source repo copied to {local_path}")
-
-
-def load_and_save_quantized_model(
-    model_name: str,
-    bits: int,
-    output_path: Path,
-    dry_run: bool = False,
-) -> None:
-    """Load a model with quantization and save it in mflux format."""
-    print(f"\n{'=' * 60}")
-    print(f"Loading {model_name} with {bits}-bit quantization...")
-    print(f"Output path: {output_path}")
-    print(f"{'=' * 60}")
-
-    if dry_run:
-        print("[DRY RUN] Would load and save quantized model")
-        return
-
-    from mflux.models.common.config.model_config import ModelConfig
-
-    model_class = get_model_class(model_name)
-    model_config = ModelConfig.from_name(model_name=model_name, base_model=None)
-
-    model: Flux1 = model_class(
-        quantize=bits,
-        model_config=model_config,
-    )
-
-    print(f"Saving model to {output_path}...")
-    model.save_model(str(output_path))
-    print(f"Model saved successfully to {output_path}")
-
-
-def copy_source_metadata(
-    source_repo: str,
-    local_path: Path,
-    dry_run: bool = False,
-) -> None:
-    """Copy metadata files (LICENSE, README, etc.) from source repo, excluding safetensors."""
-    print(f"\n{'=' * 60}")
-    print(f"Copying metadata from source repo: {source_repo}")
-    print(f"{'=' * 60}")
-
-    if dry_run:
-        print("[DRY RUN] Would download metadata files (excluding *.safetensors)")
-        return
-
-    from huggingface_hub import snapshot_download
-
-    # Download all files except safetensors to our local path
-    snapshot_download(
-        repo_id=source_repo,
-        local_dir=local_path,
-        ignore_patterns=["*.safetensors"],
-    )
-    print(f"Metadata files copied to {local_path}")
-
-
-def upload_to_huggingface(
-    local_path: Path,
-    repo_id: str,
-    dry_run: bool = False,
-    clean_remote: bool = False,
-) -> None:
-    """Upload a saved model to HuggingFace."""
-    print(f"\n{'=' * 60}")
-    print(f"Uploading to HuggingFace: {repo_id}")
-    print(f"Local path: {local_path}")
-    print(f"Clean remote first: {clean_remote}")
-    print(f"{'=' * 60}")
-
-    if dry_run:
-        print("[DRY RUN] Would upload to HuggingFace")
-        return
-
-    from huggingface_hub import HfApi
-
-    api = HfApi()
-
-    # Create the repo if it doesn't exist
-    print(f"Creating/verifying repo: {repo_id}")
-    api.create_repo(repo_id=repo_id, repo_type="model", exist_ok=True)
-
-    # Clean remote repo if requested (delete old mflux-format files)
-    if clean_remote:
-        print("Cleaning old mflux-format files from remote...")
-        try:
-            # Pattern for mflux numbered shards: <dir>/<number>.safetensors
-            numbered_pattern = re.compile(r".*/\d+\.safetensors$")
-
-            repo_files = api.list_repo_files(repo_id=repo_id, repo_type="model")
-            for file_path in repo_files:
-                # Delete numbered safetensors (mflux format) and mflux index files
-                if numbered_pattern.match(file_path) or file_path.endswith(
-                    "/model.safetensors.index.json"
-                ):
-                    print(f"  Deleting: {file_path}")
-                    api.delete_file(
-                        path_in_repo=file_path, repo_id=repo_id, repo_type="model"
-                    )
-        except Exception as e:
-            print(f"Warning: Could not clean remote files: {e}")
-
-    # Upload the folder
-    print("Uploading folder contents...")
-    api.upload_folder(
-        folder_path=str(local_path),
-        repo_id=repo_id,
-        repo_type="model",
-    )
-    print(f"Upload complete: https://huggingface.co/{repo_id}")
-
-
-def clean_local_files(local_path: Path, dry_run: bool = False) -> None:
-    """Remove local model files after upload."""
-    print(f"\nCleaning up: {local_path}")
-    if dry_run:
-        print("[DRY RUN] Would remove local files")
-        return
-
-    if local_path.exists():
-        shutil.rmtree(local_path)
-        print(f"Removed {local_path}")
-
-
-def main() -> int:
-    parser = argparse.ArgumentParser(
-        description="Download an mflux model, quantize it, and upload to HuggingFace.",
-        formatter_class=argparse.RawDescriptionHelpFormatter,
-        epilog="""
-Examples:
-    # Process all variants (base, 4-bit, 8-bit) for FLUX.1-Kontext-dev
-    python tmp/quantize_and_upload.py --model black-forest-labs/FLUX.1-Kontext-dev
-
-    # Only process 4-bit variant
-    python tmp/quantize_and_upload.py --model black-forest-labs/FLUX.1-Kontext-dev --skip-base --skip-8bit
-
-    # Save locally without uploading
-    python tmp/quantize_and_upload.py --model black-forest-labs/FLUX.1-Kontext-dev --skip-upload
-
-    # Preview what would happen
-    python tmp/quantize_and_upload.py --model black-forest-labs/FLUX.1-Kontext-dev --dry-run
-        """,
-    )
-
-    parser.add_argument(
-        "--model",
-        "-m",
-        required=True,
-        help="HuggingFace model path (e.g., black-forest-labs/FLUX.1-Kontext-dev)",
-    )
-    parser.add_argument(
-        "--output-dir",
-        type=Path,
-        default=Path("./tmp/models"),
-        help="Local directory to save models (default: ./tmp/models)",
-    )
-    parser.add_argument(
-        "--skip-base",
-        action="store_true",
-        help="Skip base model (no quantization)",
-    )
-    parser.add_argument(
-        "--skip-4bit",
-        action="store_true",
-        help="Skip 4-bit quantized model",
-    )
-    parser.add_argument(
-        "--skip-8bit",
-        action="store_true",
-        help="Skip 8-bit quantized model",
-    )
-    parser.add_argument(
-        "--skip-download",
-        action="store_true",
-        help="Skip downloading/processing, only do upload/clean operations",
-    )
-    parser.add_argument(
-        "--skip-upload",
-        action="store_true",
-        help="Only save locally, don't upload to HuggingFace",
-    )
-    parser.add_argument(
-        "--clean",
-        action="store_true",
-        help="Remove local files after upload",
-    )
-    parser.add_argument(
-        "--clean-remote",
-        action="store_true",
-        help="Delete old mflux-format files from remote repo before uploading",
-    )
-    parser.add_argument(
-        "--dry-run",
-        action="store_true",
-        help="Print actions without executing",
-    )
-
-    args = parser.parse_args()
-
-    # Determine which variants to process
-    variants: list[int | None] = []
-    if not args.skip_base:
-        variants.append(None)  # Base model (no quantization)
-    if not args.skip_4bit:
-        variants.append(4)
-    if not args.skip_8bit:
-        variants.append(8)
-
-    if not variants:
-        print("Error: All variants skipped. Nothing to do.")
-        return 1
-
-    # Create output directory
-    args.output_dir.mkdir(parents=True, exist_ok=True)
-
-    print(f"Model: {args.model}")
-    print(f"Output directory: {args.output_dir}")
-    print(
-        f"Variants to process: {['base' if v is None else f'{v}-bit' for v in variants]}"
-    )
-    print(f"Upload to HuggingFace: {not args.skip_upload}")
-    print(f"Clean after upload: {args.clean}")
-    if args.dry_run:
-        print("\n*** DRY RUN MODE - No actual changes will be made ***")
-
-    # Process each variant
-    for bits in variants:
-        local_path = get_local_path(args.output_dir, args.model, bits)
-        repo_id = get_repo_name(args.model, bits)
-
-        if not args.skip_download:
-            if bits is None:
-                # Base model: copy original HF repo structure (no mflux conversion)
-                copy_source_repo(
-                    source_repo=args.model,
-                    local_path=local_path,
-                    dry_run=args.dry_run,
-                )
-            else:
-                # Quantized model: load, quantize, and save with mflux
-                load_and_save_quantized_model(
-                    model_name=args.model,
-                    bits=bits,
-                    output_path=local_path,
-                    dry_run=args.dry_run,
-                )
-
-                # Copy metadata from source repo (LICENSE, README, etc.)
-                copy_source_metadata(
-                    source_repo=args.model,
-                    local_path=local_path,
-                    dry_run=args.dry_run,
-                )
-
-        # Upload
-        if not args.skip_upload:
-            upload_to_huggingface(
-                local_path=local_path,
-                repo_id=repo_id,
-                dry_run=args.dry_run,
-                clean_remote=args.clean_remote,
-            )
-
-            # Clean up if requested
-            if args.clean:
-                clean_local_files(local_path, dry_run=args.dry_run)
-
-    print("\n" + "=" * 60)
-    print("All done!")
-    print("=" * 60)
-
-    return 0
-
-
-if __name__ == "__main__":
-    sys.exit(main())
--- a/uv.lock
+++ b/uv.lock