vllm.v1.core.sched.request_queue ¶

FCFSRequestQueue ¶

Bases: deque[Request], RequestQueue

A first-come-first-served queue that supports deque operations.

Source code in vllm/v1/core/sched/request_queue.py

class FCFSRequestQueue(deque[Request], RequestQueue):
    """A first-come-first-served queue that supports deque operations."""

    def add_request(self, request: Request) -> None:
        """Add a request to the queue according to FCFS policy."""
        self.append(request)

    def pop_request(self) -> Request:
        """Pop a request from the queue according to FCFS policy."""
        return self.popleft()

    def peek_request(self) -> Request:
        """Peek at the next request in the queue without removing it."""
        if not self:
            raise IndexError("peek from an empty queue")
        return self[0]

    def prepend_request(self, request: Request) -> None:
        """Prepend a request to the front of the queue."""
        self.appendleft(request)

    def prepend_requests(self, requests: RequestQueue) -> None:
        """Prepend all requests from another queue to the front of this
        queue."""
        self.extendleft(reversed(requests))

    def remove_request(self, request: Request) -> None:
        """Remove a specific request from the queue."""
        self.remove(request)

    def remove_requests(self, requests: Iterable[Request]) -> None:
        """Remove multiple specific requests from the queue."""
        requests_to_remove = set(requests)
        filtered_requests = [req for req in self if req not in requests_to_remove]
        # deque does not support in-place filtering, so we need to clear
        # and extend
        self.clear()
        self.extend(filtered_requests)

    def __bool__(self) -> bool:
        """Check if queue has any requests."""
        return len(self) > 0

    def __len__(self) -> int:
        """Get number of requests in queue."""
        return super().__len__()

    def __iter__(self) -> Iterator[Request]:
        """Iterate over the queue according to FCFS policy."""
        return super().__iter__()

    def __reversed__(self) -> Iterator[Request]:
        """Iterate over the queue in reverse order."""
        return super().__reversed__()

bool ¶

__bool__() -> bool

Check if queue has any requests.

Source code in vllm/v1/core/sched/request_queue.py

def __bool__(self) -> bool:
    """Check if queue has any requests."""
    return len(self) > 0

iter ¶

__iter__() -> Iterator[Request]

Iterate over the queue according to FCFS policy.

Source code in vllm/v1/core/sched/request_queue.py

def __iter__(self) -> Iterator[Request]:
    """Iterate over the queue according to FCFS policy."""
    return super().__iter__()

len ¶

__len__() -> int

Get number of requests in queue.

Source code in vllm/v1/core/sched/request_queue.py

def __len__(self) -> int:
    """Get number of requests in queue."""
    return super().__len__()

reversed ¶

__reversed__() -> Iterator[Request]

Iterate over the queue in reverse order.

Source code in vllm/v1/core/sched/request_queue.py

def __reversed__(self) -> Iterator[Request]:
    """Iterate over the queue in reverse order."""
    return super().__reversed__()

add_request ¶

add_request(request: Request) -> None

Add a request to the queue according to FCFS policy.

Source code in vllm/v1/core/sched/request_queue.py

def add_request(self, request: Request) -> None:
    """Add a request to the queue according to FCFS policy."""
    self.append(request)

peek_request ¶

peek_request() -> Request

Peek at the next request in the queue without removing it.

Source code in vllm/v1/core/sched/request_queue.py

def peek_request(self) -> Request:
    """Peek at the next request in the queue without removing it."""
    if not self:
        raise IndexError("peek from an empty queue")
    return self[0]

pop_request ¶

pop_request() -> Request

Pop a request from the queue according to FCFS policy.

Source code in vllm/v1/core/sched/request_queue.py

def pop_request(self) -> Request:
    """Pop a request from the queue according to FCFS policy."""
    return self.popleft()

prepend_request ¶

prepend_request(request: Request) -> None

Prepend a request to the front of the queue.

Source code in vllm/v1/core/sched/request_queue.py

def prepend_request(self, request: Request) -> None:
    """Prepend a request to the front of the queue."""
    self.appendleft(request)

prepend_requests ¶

prepend_requests(requests: RequestQueue) -> None

Prepend all requests from another queue to the front of this queue.

Source code in vllm/v1/core/sched/request_queue.py

def prepend_requests(self, requests: RequestQueue) -> None:
    """Prepend all requests from another queue to the front of this
    queue."""
    self.extendleft(reversed(requests))

remove_request ¶

remove_request(request: Request) -> None

Remove a specific request from the queue.

Source code in vllm/v1/core/sched/request_queue.py

def remove_request(self, request: Request) -> None:
    """Remove a specific request from the queue."""
    self.remove(request)

remove_requests ¶

remove_requests(requests: Iterable[Request]) -> None

Remove multiple specific requests from the queue.

Source code in vllm/v1/core/sched/request_queue.py

def remove_requests(self, requests: Iterable[Request]) -> None:
    """Remove multiple specific requests from the queue."""
    requests_to_remove = set(requests)
    filtered_requests = [req for req in self if req not in requests_to_remove]
    # deque does not support in-place filtering, so we need to clear
    # and extend
    self.clear()
    self.extend(filtered_requests)

PriorityRequestQueue ¶

Bases: RequestQueue

A priority queue that supports heap operations.

Requests with a smaller value of priority are processed first. If multiple requests have the same priority, the one with the earlier arrival_time is processed first.

Source code in vllm/v1/core/sched/request_queue.py

class PriorityRequestQueue(RequestQueue):
    """
    A priority queue that supports heap operations.

    Requests with a smaller value of `priority` are processed first.
    If multiple requests have the same priority, the one with the earlier
    `arrival_time` is processed first.
    """

    def __init__(self) -> None:
        self._heap: list[tuple[int, float, Request]] = []

    def add_request(self, request: Request) -> None:
        """Add a request to the queue according to priority policy."""
        heapq.heappush(self._heap, (request.priority, request.arrival_time, request))

    def pop_request(self) -> Request:
        """Pop a request from the queue according to priority policy."""
        if not self._heap:
            raise IndexError("pop from empty heap")
        _, _, request = heapq.heappop(self._heap)
        return request

    def peek_request(self) -> Request:
        """Peek at the next request in the queue without removing it."""
        if not self._heap:
            raise IndexError("peek from empty heap")
        _, _, request = self._heap[0]
        return request

    def prepend_request(self, request: Request) -> None:
        """Add a request to the queue according to priority policy.

        Note: In a priority queue, there is no concept of prepending to the
        front. Requests are ordered by (priority, arrival_time)."""
        self.add_request(request)

    def prepend_requests(self, requests: RequestQueue) -> None:
        """Add all requests from another queue according to priority policy.

        Note: In a priority queue, there is no concept of prepending to the
        front. Requests are ordered by (priority, arrival_time)."""
        for request in requests:
            self.add_request(request)

    def remove_request(self, request: Request) -> None:
        """Remove a specific request from the queue."""
        self._heap = [(p, t, r) for p, t, r in self._heap if r != request]
        heapq.heapify(self._heap)

    def remove_requests(self, requests: Iterable[Request]) -> None:
        """Remove multiple specific requests from the queue."""
        requests_to_remove = set(requests)
        self._heap = [
            (p, t, r) for p, t, r in self._heap if r not in requests_to_remove
        ]
        heapq.heapify(self._heap)

    def __bool__(self) -> bool:
        """Check if queue has any requests."""
        return bool(self._heap)

    def __len__(self) -> int:
        """Get number of requests in queue."""
        return len(self._heap)

    def __iter__(self) -> Iterator[Request]:
        """Iterate over the queue according to priority policy."""
        heap_copy = self._heap[:]
        while heap_copy:
            _, _, request = heapq.heappop(heap_copy)
            yield request

    def __reversed__(self) -> Iterator[Request]:
        """Iterate over the queue in reverse priority order."""
        return reversed(list(self))

_heap `instance-attribute` ¶

_heap: list[tuple[int, float, Request]] = []

bool ¶

__bool__() -> bool

Check if queue has any requests.

Source code in vllm/v1/core/sched/request_queue.py

def __bool__(self) -> bool:
    """Check if queue has any requests."""
    return bool(self._heap)

init ¶

__init__() -> None

Source code in vllm/v1/core/sched/request_queue.py

def __init__(self) -> None:
    self._heap: list[tuple[int, float, Request]] = []

iter ¶

__iter__() -> Iterator[Request]

Iterate over the queue according to priority policy.

Source code in vllm/v1/core/sched/request_queue.py

def __iter__(self) -> Iterator[Request]:
    """Iterate over the queue according to priority policy."""
    heap_copy = self._heap[:]
    while heap_copy:
        _, _, request = heapq.heappop(heap_copy)
        yield request

len ¶

__len__() -> int

Get number of requests in queue.

Source code in vllm/v1/core/sched/request_queue.py

def __len__(self) -> int:
    """Get number of requests in queue."""
    return len(self._heap)

reversed ¶

__reversed__() -> Iterator[Request]

Iterate over the queue in reverse priority order.

Source code in vllm/v1/core/sched/request_queue.py

def __reversed__(self) -> Iterator[Request]:
    """Iterate over the queue in reverse priority order."""
    return reversed(list(self))

add_request ¶

add_request(request: Request) -> None

Add a request to the queue according to priority policy.

Source code in vllm/v1/core/sched/request_queue.py

def add_request(self, request: Request) -> None:
    """Add a request to the queue according to priority policy."""
    heapq.heappush(self._heap, (request.priority, request.arrival_time, request))

peek_request ¶

peek_request() -> Request

Peek at the next request in the queue without removing it.

Source code in vllm/v1/core/sched/request_queue.py

def peek_request(self) -> Request:
    """Peek at the next request in the queue without removing it."""
    if not self._heap:
        raise IndexError("peek from empty heap")
    _, _, request = self._heap[0]
    return request

pop_request ¶

pop_request() -> Request

Pop a request from the queue according to priority policy.

Source code in vllm/v1/core/sched/request_queue.py

def pop_request(self) -> Request:
    """Pop a request from the queue according to priority policy."""
    if not self._heap:
        raise IndexError("pop from empty heap")
    _, _, request = heapq.heappop(self._heap)
    return request

prepend_request ¶

prepend_request(request: Request) -> None

Add a request to the queue according to priority policy.

Note: In a priority queue, there is no concept of prepending to the front. Requests are ordered by (priority, arrival_time).

Source code in vllm/v1/core/sched/request_queue.py

def prepend_request(self, request: Request) -> None:
    """Add a request to the queue according to priority policy.

    Note: In a priority queue, there is no concept of prepending to the
    front. Requests are ordered by (priority, arrival_time)."""
    self.add_request(request)

prepend_requests ¶

prepend_requests(requests: RequestQueue) -> None

Add all requests from another queue according to priority policy.

Note: In a priority queue, there is no concept of prepending to the front. Requests are ordered by (priority, arrival_time).

Source code in vllm/v1/core/sched/request_queue.py

def prepend_requests(self, requests: RequestQueue) -> None:
    """Add all requests from another queue according to priority policy.

    Note: In a priority queue, there is no concept of prepending to the
    front. Requests are ordered by (priority, arrival_time)."""
    for request in requests:
        self.add_request(request)

remove_request ¶

remove_request(request: Request) -> None

Remove a specific request from the queue.

Source code in vllm/v1/core/sched/request_queue.py

def remove_request(self, request: Request) -> None:
    """Remove a specific request from the queue."""
    self._heap = [(p, t, r) for p, t, r in self._heap if r != request]
    heapq.heapify(self._heap)

remove_requests ¶

remove_requests(requests: Iterable[Request]) -> None

Remove multiple specific requests from the queue.

Source code in vllm/v1/core/sched/request_queue.py

def remove_requests(self, requests: Iterable[Request]) -> None:
    """Remove multiple specific requests from the queue."""
    requests_to_remove = set(requests)
    self._heap = [
        (p, t, r) for p, t, r in self._heap if r not in requests_to_remove
    ]
    heapq.heapify(self._heap)

RequestQueue ¶

Bases: ABC

Abstract base class for request queues.

Source code in vllm/v1/core/sched/request_queue.py

class RequestQueue(ABC):
    """Abstract base class for request queues."""

    @abstractmethod
    def add_request(self, request: Request) -> None:
        """Add a request to the queue according to the policy."""
        pass

    @abstractmethod
    def pop_request(self) -> Request:
        """Pop a request from the queue according to the policy."""
        pass

    @abstractmethod
    def peek_request(self) -> Request:
        """Peek at the request at the front of the queue without removing it."""
        pass

    @abstractmethod
    def prepend_request(self, request: Request) -> None:
        """Prepend a request to the front of the queue."""
        pass

    @abstractmethod
    def prepend_requests(self, requests: "RequestQueue") -> None:
        """Prepend all requests from another queue to the front of this
        queue."""
        pass

    @abstractmethod
    def remove_request(self, request: Request) -> None:
        """Remove a specific request from the queue."""
        pass

    @abstractmethod
    def remove_requests(self, requests: Iterable[Request]) -> None:
        """Remove multiple specific requests from the queue."""
        pass

    @abstractmethod
    def __bool__(self) -> bool:
        """Check if queue has any requests."""
        pass

    @abstractmethod
    def __len__(self) -> int:
        """Get number of requests in queue."""
        pass

    @abstractmethod
    def __iter__(self) -> Iterator[Request]:
        """Iterate over the queue according to the policy."""
        pass

    @abstractmethod
    def __reversed__(self) -> Iterator[Request]:
        """Iterate over the queue in reverse order."""
        pass

bool `abstractmethod` ¶

__bool__() -> bool

Check if queue has any requests.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def __bool__(self) -> bool:
    """Check if queue has any requests."""
    pass

iter `abstractmethod` ¶

__iter__() -> Iterator[Request]

Iterate over the queue according to the policy.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def __iter__(self) -> Iterator[Request]:
    """Iterate over the queue according to the policy."""
    pass

len `abstractmethod` ¶

__len__() -> int

Get number of requests in queue.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def __len__(self) -> int:
    """Get number of requests in queue."""
    pass

reversed `abstractmethod` ¶

__reversed__() -> Iterator[Request]

Iterate over the queue in reverse order.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def __reversed__(self) -> Iterator[Request]:
    """Iterate over the queue in reverse order."""
    pass

add_request `abstractmethod` ¶

add_request(request: Request) -> None

Add a request to the queue according to the policy.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def add_request(self, request: Request) -> None:
    """Add a request to the queue according to the policy."""
    pass

peek_request `abstractmethod` ¶

peek_request() -> Request

Peek at the request at the front of the queue without removing it.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def peek_request(self) -> Request:
    """Peek at the request at the front of the queue without removing it."""
    pass

pop_request `abstractmethod` ¶

pop_request() -> Request

Pop a request from the queue according to the policy.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def pop_request(self) -> Request:
    """Pop a request from the queue according to the policy."""
    pass

prepend_request `abstractmethod` ¶

prepend_request(request: Request) -> None

Prepend a request to the front of the queue.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def prepend_request(self, request: Request) -> None:
    """Prepend a request to the front of the queue."""
    pass

prepend_requests `abstractmethod` ¶

prepend_requests(requests: RequestQueue) -> None

Prepend all requests from another queue to the front of this queue.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def prepend_requests(self, requests: "RequestQueue") -> None:
    """Prepend all requests from another queue to the front of this
    queue."""
    pass

remove_request `abstractmethod` ¶

remove_request(request: Request) -> None

Remove a specific request from the queue.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def remove_request(self, request: Request) -> None:
    """Remove a specific request from the queue."""
    pass

remove_requests `abstractmethod` ¶

remove_requests(requests: Iterable[Request]) -> None

Remove multiple specific requests from the queue.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def remove_requests(self, requests: Iterable[Request]) -> None:
    """Remove multiple specific requests from the queue."""
    pass

SchedulingPolicy ¶

Bases: Enum

Enum for scheduling policies.

Source code in vllm/v1/core/sched/request_queue.py

class SchedulingPolicy(Enum):
    """Enum for scheduling policies."""

    FCFS = "fcfs"
    PRIORITY = "priority"

FCFS `class-attribute` `instance-attribute` ¶

FCFS = 'fcfs'

PRIORITY `class-attribute` `instance-attribute` ¶

PRIORITY = 'priority'

create_request_queue ¶

create_request_queue(
    policy: SchedulingPolicy,
) -> RequestQueue

Create request queue based on scheduling policy.

Source code in vllm/v1/core/sched/request_queue.py

def create_request_queue(policy: SchedulingPolicy) -> RequestQueue:
    """Create request queue based on scheduling policy."""
    if policy == SchedulingPolicy.PRIORITY:
        return PriorityRequestQueue()
    elif policy == SchedulingPolicy.FCFS:
        return FCFSRequestQueue()
    else:
        raise ValueError(f"Unknown scheduling policy: {policy}")

vllm.v1.core.sched.request_queue ¶

FCFSRequestQueue ¶

__bool__ ¶

__iter__ ¶

__len__ ¶

__reversed__ ¶

add_request ¶

peek_request ¶

pop_request ¶

prepend_request ¶

prepend_requests ¶

remove_request ¶

remove_requests ¶

PriorityRequestQueue ¶

_heap instance-attribute ¶

__bool__ ¶

__init__ ¶

__iter__ ¶

__len__ ¶

__reversed__ ¶

add_request ¶

peek_request ¶

pop_request ¶

prepend_request ¶

prepend_requests ¶

remove_request ¶

remove_requests ¶

RequestQueue ¶

__bool__ abstractmethod ¶

__iter__ abstractmethod ¶

__len__ abstractmethod ¶

__reversed__ abstractmethod ¶

add_request abstractmethod ¶

peek_request abstractmethod ¶

pop_request abstractmethod ¶

prepend_request abstractmethod ¶

prepend_requests abstractmethod ¶

remove_request abstractmethod ¶

remove_requests abstractmethod ¶

SchedulingPolicy ¶

FCFS class-attribute instance-attribute ¶

PRIORITY class-attribute instance-attribute ¶

create_request_queue ¶

bool ¶

iter ¶

len ¶

reversed ¶

_heap `instance-attribute` ¶

bool ¶

init ¶

iter ¶

len ¶

reversed ¶

bool `abstractmethod` ¶

iter `abstractmethod` ¶

len `abstractmethod` ¶

reversed `abstractmethod` ¶

add_request `abstractmethod` ¶

peek_request `abstractmethod` ¶

pop_request `abstractmethod` ¶

prepend_request `abstractmethod` ¶

prepend_requests `abstractmethod` ¶

remove_request `abstractmethod` ¶

remove_requests `abstractmethod` ¶

FCFS `class-attribute` `instance-attribute` ¶

PRIORITY `class-attribute` `instance-attribute` ¶