Skip to content

vllm.config.engram

Classes:

  • EngramConfig –

    Configuration for Engram embedding storage and sharding.

Functions:

EngramConfig

Configuration for Engram embedding storage and sharding.

Methods:

Attributes:

Source code in vllm/config/engram.py
@config
class EngramConfig:
    """Configuration for Engram embedding storage and sharding."""

    cpu_offload: bool = Field(default_factory=_default_cpu_offload)
    """Store embedding weights in pinned CPU memory for UVA lookup.
    Defaults to VLLM_PLE_CPU_OFFLOAD, which is enabled by default. An explicit
    value takes precedence over the environment variable."""

    embedding_across_dp: bool = False
    """Shard embeddings across TP and all DP ranks when enabled.
    Otherwise, each DP rank has a separate TP-sharded embedding replica."""

    dp_shared_memory: bool = False
    """Share CPU-offloaded embedding weights between co-located
    DP replicas. Each node stores one copy of every TP shard, reducing host
    memory without per-step Engram DP collectives. Requires sufficient
    /dev/shm capacity and a shared IPC namespace."""

    @model_validator(mode="after")
    def _validate_shared_memory(self) -> Self:
        if self.dp_shared_memory and not self.cpu_offload:
            raise ValueError("dp_shared_memory requires cpu_offload=True")
        return self

    def verify_model_config(self, model_config: "ModelConfig | None") -> None:
        """Reject Engram configuration for models without n-gram embeddings."""
        from vllm.platforms import current_platform

        field = (
            _NGRAM_LAYER_FIELDS.get(model_config.architecture)
            if model_config is not None
            else None
        )
        if (
            model_config is None
            or field is None
            or not current_platform.is_cuda()
            or not getattr(model_config.hf_text_config, field, None)
        ):
            raise ValueError(
                "EngramConfig requires a model with supported Engram "
                "embeddings, non-empty n-gram layer ids, and CUDA."
            )

    def verify_parallel_config(self, parallel_config: "ParallelConfig") -> None:
        """Reject unsupported embedding parallel topologies."""
        if self.dp_shared_memory:
            if parallel_config.data_parallel_size <= 1:
                raise ValueError("dp_shared_memory requires data_parallel_size > 1.")
            if parallel_config.enable_elastic_ep:
                raise ValueError("dp_shared_memory is not supported with elastic EP.")
        if (
            self.embedding_across_dp
            and parallel_config.data_parallel_size > 1
            and parallel_config.enable_elastic_ep
        ):
            raise ValueError(
                "Engram embedding_across_dp is not supported with elastic EP yet."
            )

    def verify_load_config(self, load_config: "LoadConfig") -> None:
        """Shared tables require a loader that invokes parameter weight callbacks."""
        if self.dp_shared_memory and load_config.load_format not in (
            "auto",
            "safetensors",
            "pt",
        ):
            raise ValueError(
                "dp_shared_memory requires load_format 'auto', "
                f"'safetensors' or 'pt'; got {load_config.load_format!r}."
            )

    def get_parallel_size(self, parallel_config: "ParallelConfig") -> int:
        """Derive the embedding group size from the parallel configuration."""
        size = parallel_config.tensor_parallel_size
        if self.embedding_across_dp and parallel_config.data_parallel_size > 1:
            size *= parallel_config.data_parallel_size
        return size

    def compute_hash(self) -> str:
        """Hash settings that affect embedding execution and graph structure."""
        return hash_factors(get_hash_factors(self, set()))

cpu_offload = Field(default_factory=_default_cpu_offload) class-attribute instance-attribute

Store embedding weights in pinned CPU memory for UVA lookup. Defaults to VLLM_PLE_CPU_OFFLOAD, which is enabled by default. An explicit value takes precedence over the environment variable.

dp_shared_memory = False class-attribute instance-attribute

Share CPU-offloaded embedding weights between co-located DP replicas. Each node stores one copy of every TP shard, reducing host memory without per-step Engram DP collectives. Requires sufficient /dev/shm capacity and a shared IPC namespace.

embedding_across_dp = False class-attribute instance-attribute

Shard embeddings across TP and all DP ranks when enabled. Otherwise, each DP rank has a separate TP-sharded embedding replica.

compute_hash()

Hash settings that affect embedding execution and graph structure.

Source code in vllm/config/engram.py
def compute_hash(self) -> str:
    """Hash settings that affect embedding execution and graph structure."""
    return hash_factors(get_hash_factors(self, set()))

get_parallel_size(parallel_config)

Derive the embedding group size from the parallel configuration.

Source code in vllm/config/engram.py
def get_parallel_size(self, parallel_config: "ParallelConfig") -> int:
    """Derive the embedding group size from the parallel configuration."""
    size = parallel_config.tensor_parallel_size
    if self.embedding_across_dp and parallel_config.data_parallel_size > 1:
        size *= parallel_config.data_parallel_size
    return size

verify_load_config(load_config)

Shared tables require a loader that invokes parameter weight callbacks.

Source code in vllm/config/engram.py
def verify_load_config(self, load_config: "LoadConfig") -> None:
    """Shared tables require a loader that invokes parameter weight callbacks."""
    if self.dp_shared_memory and load_config.load_format not in (
        "auto",
        "safetensors",
        "pt",
    ):
        raise ValueError(
            "dp_shared_memory requires load_format 'auto', "
            f"'safetensors' or 'pt'; got {load_config.load_format!r}."
        )

verify_model_config(model_config)

Reject Engram configuration for models without n-gram embeddings.

Source code in vllm/config/engram.py
def verify_model_config(self, model_config: "ModelConfig | None") -> None:
    """Reject Engram configuration for models without n-gram embeddings."""
    from vllm.platforms import current_platform

    field = (
        _NGRAM_LAYER_FIELDS.get(model_config.architecture)
        if model_config is not None
        else None
    )
    if (
        model_config is None
        or field is None
        or not current_platform.is_cuda()
        or not getattr(model_config.hf_text_config, field, None)
    ):
        raise ValueError(
            "EngramConfig requires a model with supported Engram "
            "embeddings, non-empty n-gram layer ids, and CUDA."
        )

verify_parallel_config(parallel_config)

Reject unsupported embedding parallel topologies.

Source code in vllm/config/engram.py
def verify_parallel_config(self, parallel_config: "ParallelConfig") -> None:
    """Reject unsupported embedding parallel topologies."""
    if self.dp_shared_memory:
        if parallel_config.data_parallel_size <= 1:
            raise ValueError("dp_shared_memory requires data_parallel_size > 1.")
        if parallel_config.enable_elastic_ep:
            raise ValueError("dp_shared_memory is not supported with elastic EP.")
    if (
        self.embedding_across_dp
        and parallel_config.data_parallel_size > 1
        and parallel_config.enable_elastic_ep
    ):
        raise ValueError(
            "Engram embedding_across_dp is not supported with elastic EP yet."
        )

model_has_engram_layers(model_config)

Whether the model carries n-gram embedding layers.

Source code in vllm/config/engram.py
def model_has_engram_layers(model_config: "ModelConfig | None") -> bool:
    """Whether the model carries n-gram embedding layers."""
    if model_config is None:
        return False
    field = _NGRAM_LAYER_FIELDS.get(model_config.architecture)
    if field is None:
        return False
    return bool(getattr(model_config.hf_text_config, field, None))