`vllm.v1.core.sched.request_queue` ¶

Classes:

FCFSRequestQueue –

A first-come-first-served queue that supports deque operations.
PriorityRequestQueue –

A priority queue that supports heap operations.
RequestQueue –

Abstract base class for request queues.
SchedulingPolicy –

Enum for scheduling policies.

Functions:

create_request_queue –

Create request queue based on scheduling policy.

`FCFSRequestQueue` ¶

Bases: deque[Request], RequestQueue

A first-come-first-served queue that supports deque operations.

Methods:

__bool__ –

Check if queue has any requests.
__iter__ –

Iterate over the queue according to FCFS policy.
__len__ –

Get number of requests in queue.
add_request –

Add a request to the queue according to FCFS policy.
peek_request –

Peek at the next request in the queue without removing it.
pop_request –

Pop a request from the queue according to FCFS policy.
prepend_request –

Prepend a request to the front of the queue.
prepend_requests –

Prepend all requests from another queue to the front of this
remove_request –

Remove a specific request from the queue.
remove_requests –

Remove multiple specific requests from the queue.

Source code in vllm/v1/core/sched/request_queue.py

class FCFSRequestQueue(deque[Request], RequestQueue):
    """A first-come-first-served queue that supports deque operations."""

    def add_request(self, request: Request) -> None:
        """Add a request to the queue according to FCFS policy."""
        self.append(request)

    def pop_request(self) -> Request:
        """Pop a request from the queue according to FCFS policy."""
        return self.popleft()

    def peek_request(self) -> Request:
        """Peek at the next request in the queue without removing it."""
        if not self:
            raise IndexError("peek from an empty queue")
        return self[0]

    def prepend_request(self, request: Request) -> None:
        """Prepend a request to the front of the queue."""
        self.appendleft(request)

    def prepend_requests(self, requests: RequestQueue) -> None:
        """Prepend all requests from another queue to the front of this
        queue.

        Note: The requests will be prepended in reverse order of their
        appearance in the `requests` queue.
        """
        self.extendleft(requests)

    def remove_request(self, request: Request) -> None:
        """Remove a specific request from the queue."""
        self.remove(request)

    def remove_requests(self, requests: Iterable[Request]) -> None:
        """Remove multiple specific requests from the queue."""
        requests_to_remove = set(requests)
        filtered_requests = [req for req in self if req not in requests_to_remove]
        # deque does not support in-place filtering, so we need to clear
        # and extend
        self.clear()
        self.extend(filtered_requests)

    def __bool__(self) -> bool:
        """Check if queue has any requests."""
        return len(self) > 0

    def __len__(self) -> int:
        """Get number of requests in queue."""
        return super().__len__()

    def __iter__(self) -> Iterator[Request]:
        """Iterate over the queue according to FCFS policy."""
        return super().__iter__()

`bool()` ¶

Check if queue has any requests.

Source code in vllm/v1/core/sched/request_queue.py

def __bool__(self) -> bool:
    """Check if queue has any requests."""
    return len(self) > 0

`iter()` ¶

Iterate over the queue according to FCFS policy.

Source code in vllm/v1/core/sched/request_queue.py

def __iter__(self) -> Iterator[Request]:
    """Iterate over the queue according to FCFS policy."""
    return super().__iter__()

`len()` ¶

Get number of requests in queue.

Source code in vllm/v1/core/sched/request_queue.py

def __len__(self) -> int:
    """Get number of requests in queue."""
    return super().__len__()

`add_request(request)` ¶

Add a request to the queue according to FCFS policy.

Source code in vllm/v1/core/sched/request_queue.py

def add_request(self, request: Request) -> None:
    """Add a request to the queue according to FCFS policy."""
    self.append(request)

`peek_request()` ¶

Peek at the next request in the queue without removing it.

Source code in vllm/v1/core/sched/request_queue.py

def peek_request(self) -> Request:
    """Peek at the next request in the queue without removing it."""
    if not self:
        raise IndexError("peek from an empty queue")
    return self[0]

`pop_request()` ¶

Pop a request from the queue according to FCFS policy.

Source code in vllm/v1/core/sched/request_queue.py

def pop_request(self) -> Request:
    """Pop a request from the queue according to FCFS policy."""
    return self.popleft()

`prepend_request(request)` ¶

Prepend a request to the front of the queue.

Source code in vllm/v1/core/sched/request_queue.py

def prepend_request(self, request: Request) -> None:
    """Prepend a request to the front of the queue."""
    self.appendleft(request)

`prepend_requests(requests)` ¶

Prepend all requests from another queue to the front of this queue.

Note: The requests will be prepended in reverse order of their appearance in the requests queue.

Source code in vllm/v1/core/sched/request_queue.py

def prepend_requests(self, requests: RequestQueue) -> None:
    """Prepend all requests from another queue to the front of this
    queue.

    Note: The requests will be prepended in reverse order of their
    appearance in the `requests` queue.
    """
    self.extendleft(requests)

`remove_request(request)` ¶

Remove a specific request from the queue.

Source code in vllm/v1/core/sched/request_queue.py

def remove_request(self, request: Request) -> None:
    """Remove a specific request from the queue."""
    self.remove(request)

`remove_requests(requests)` ¶

Remove multiple specific requests from the queue.

Source code in vllm/v1/core/sched/request_queue.py

def remove_requests(self, requests: Iterable[Request]) -> None:
    """Remove multiple specific requests from the queue."""
    requests_to_remove = set(requests)
    filtered_requests = [req for req in self if req not in requests_to_remove]
    # deque does not support in-place filtering, so we need to clear
    # and extend
    self.clear()
    self.extend(filtered_requests)

`PriorityRequestQueue` ¶

Bases: RequestQueue

A priority queue that supports heap operations.

Respects the ordering defined in the Request class, where requests with a smaller value of priority are processed first. If multiple requests have the same priority, the one with the earlier arrival_time is processed first.

Methods:

__bool__ –

Check if queue has any requests.
__iter__ –

Iterate over the queue according to priority policy.
__len__ –

Get number of requests in queue.
add_request –

Add a request to the queue according to priority policy.
peek_request –

Peek at the next request in the queue without removing it.
pop_request –

Pop a request from the queue according to priority policy.
prepend_request –

Add a request to the queue according to priority policy.
prepend_requests –

Add all requests from another queue according to priority policy.
remove_request –

Remove a specific request from the queue.
remove_requests –

Remove multiple specific requests from the queue.

Source code in vllm/v1/core/sched/request_queue.py

class PriorityRequestQueue(RequestQueue):
    """
    A priority queue that supports heap operations.

    Respects the ordering defined in the Request class, where
    requests with a smaller value of `priority` are processed first.
    If multiple requests have the same priority, the one with the earlier
    `arrival_time` is processed first.
    """

    def __init__(self) -> None:
        self._heap: list[Request] = []

    def add_request(self, request: Request) -> None:
        """Add a request to the queue according to priority policy."""
        heapq.heappush(self._heap, request)

    def pop_request(self) -> Request:
        """Pop a request from the queue according to priority policy."""
        if not self._heap:
            raise IndexError("pop from empty heap")
        return heapq.heappop(self._heap)

    def peek_request(self) -> Request:
        """Peek at the next request in the queue without removing it."""
        if not self._heap:
            raise IndexError("peek from empty heap")
        return self._heap[0]

    def prepend_request(self, request: Request) -> None:
        """Add a request to the queue according to priority policy.

        Note: In a priority queue, there is no concept of prepending to the
        front. Requests are ordered by (priority, arrival_time)."""
        self.add_request(request)

    def prepend_requests(self, requests: RequestQueue) -> None:
        """Add all requests from another queue according to priority policy.

        Note: In a priority queue, there is no concept of prepending to the
        front. Requests are ordered by (priority, arrival_time)."""
        for request in requests:
            self.add_request(request)

    def remove_request(self, request: Request) -> None:
        """Remove a specific request from the queue."""
        self._heap.remove(request)
        heapq.heapify(self._heap)

    def remove_requests(self, requests: Iterable[Request]) -> None:
        """Remove multiple specific requests from the queue."""
        requests_to_remove = requests if isinstance(requests, set) else set(requests)
        self._heap = [r for r in self._heap if r not in requests_to_remove]
        heapq.heapify(self._heap)

    def __bool__(self) -> bool:
        """Check if queue has any requests."""
        return bool(self._heap)

    def __len__(self) -> int:
        """Get number of requests in queue."""
        return len(self._heap)

    def __iter__(self) -> Iterator[Request]:
        """Iterate over the queue according to priority policy."""
        heap_copy = self._heap[:]
        while heap_copy:
            yield heapq.heappop(heap_copy)

`bool()` ¶

Check if queue has any requests.

Source code in vllm/v1/core/sched/request_queue.py

def __bool__(self) -> bool:
    """Check if queue has any requests."""
    return bool(self._heap)

`iter()` ¶

Iterate over the queue according to priority policy.

Source code in vllm/v1/core/sched/request_queue.py

def __iter__(self) -> Iterator[Request]:
    """Iterate over the queue according to priority policy."""
    heap_copy = self._heap[:]
    while heap_copy:
        yield heapq.heappop(heap_copy)

`len()` ¶

Get number of requests in queue.

Source code in vllm/v1/core/sched/request_queue.py

def __len__(self) -> int:
    """Get number of requests in queue."""
    return len(self._heap)

`add_request(request)` ¶

Add a request to the queue according to priority policy.

Source code in vllm/v1/core/sched/request_queue.py

def add_request(self, request: Request) -> None:
    """Add a request to the queue according to priority policy."""
    heapq.heappush(self._heap, request)

`peek_request()` ¶

Peek at the next request in the queue without removing it.

Source code in vllm/v1/core/sched/request_queue.py

def peek_request(self) -> Request:
    """Peek at the next request in the queue without removing it."""
    if not self._heap:
        raise IndexError("peek from empty heap")
    return self._heap[0]

`pop_request()` ¶

Pop a request from the queue according to priority policy.

Source code in vllm/v1/core/sched/request_queue.py

def pop_request(self) -> Request:
    """Pop a request from the queue according to priority policy."""
    if not self._heap:
        raise IndexError("pop from empty heap")
    return heapq.heappop(self._heap)

`prepend_request(request)` ¶

Add a request to the queue according to priority policy.

Note: In a priority queue, there is no concept of prepending to the front. Requests are ordered by (priority, arrival_time).

Source code in vllm/v1/core/sched/request_queue.py

def prepend_request(self, request: Request) -> None:
    """Add a request to the queue according to priority policy.

    Note: In a priority queue, there is no concept of prepending to the
    front. Requests are ordered by (priority, arrival_time)."""
    self.add_request(request)

`prepend_requests(requests)` ¶

Add all requests from another queue according to priority policy.

Note: In a priority queue, there is no concept of prepending to the front. Requests are ordered by (priority, arrival_time).

Source code in vllm/v1/core/sched/request_queue.py

def prepend_requests(self, requests: RequestQueue) -> None:
    """Add all requests from another queue according to priority policy.

    Note: In a priority queue, there is no concept of prepending to the
    front. Requests are ordered by (priority, arrival_time)."""
    for request in requests:
        self.add_request(request)

`remove_request(request)` ¶

Remove a specific request from the queue.

Source code in vllm/v1/core/sched/request_queue.py

def remove_request(self, request: Request) -> None:
    """Remove a specific request from the queue."""
    self._heap.remove(request)
    heapq.heapify(self._heap)

`remove_requests(requests)` ¶

Remove multiple specific requests from the queue.

Source code in vllm/v1/core/sched/request_queue.py

def remove_requests(self, requests: Iterable[Request]) -> None:
    """Remove multiple specific requests from the queue."""
    requests_to_remove = requests if isinstance(requests, set) else set(requests)
    self._heap = [r for r in self._heap if r not in requests_to_remove]
    heapq.heapify(self._heap)

`RequestQueue` ¶

Bases: ABC

Abstract base class for request queues.

Methods:

__bool__ –

Check if queue has any requests.
__iter__ –

Iterate over the queue according to the policy.
__len__ –

Get number of requests in queue.
add_request –

Add a request to the queue according to the policy.
peek_request –

Peek at the request at the front of the queue without removing it.
pop_request –

Pop a request from the queue according to the policy.
prepend_request –

Prepend a request to the front of the queue.
prepend_requests –

Prepend all requests from another queue to the front of this
remove_request –

Remove a specific request from the queue.
remove_requests –

Remove multiple specific requests from the queue.

Source code in vllm/v1/core/sched/request_queue.py

class RequestQueue(ABC):
    """Abstract base class for request queues."""

    @abstractmethod
    def add_request(self, request: Request) -> None:
        """Add a request to the queue according to the policy."""
        pass

    @abstractmethod
    def pop_request(self) -> Request:
        """Pop a request from the queue according to the policy."""
        pass

    @abstractmethod
    def peek_request(self) -> Request:
        """Peek at the request at the front of the queue without removing it."""
        pass

    @abstractmethod
    def prepend_request(self, request: Request) -> None:
        """Prepend a request to the front of the queue."""
        pass

    @abstractmethod
    def prepend_requests(self, requests: "RequestQueue") -> None:
        """Prepend all requests from another queue to the front of this
        queue."""
        pass

    @abstractmethod
    def remove_request(self, request: Request) -> None:
        """Remove a specific request from the queue."""
        pass

    @abstractmethod
    def remove_requests(self, requests: Iterable[Request]) -> None:
        """Remove multiple specific requests from the queue."""
        pass

    @abstractmethod
    def __bool__(self) -> bool:
        """Check if queue has any requests."""
        pass

    @abstractmethod
    def __len__(self) -> int:
        """Get number of requests in queue."""
        pass

    @abstractmethod
    def __iter__(self) -> Iterator[Request]:
        """Iterate over the queue according to the policy."""
        pass

`bool()` `abstractmethod` ¶

Check if queue has any requests.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def __bool__(self) -> bool:
    """Check if queue has any requests."""
    pass

`iter()` `abstractmethod` ¶

Iterate over the queue according to the policy.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def __iter__(self) -> Iterator[Request]:
    """Iterate over the queue according to the policy."""
    pass

`len()` `abstractmethod` ¶

Get number of requests in queue.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def __len__(self) -> int:
    """Get number of requests in queue."""
    pass

`add_request(request)` `abstractmethod` ¶

Add a request to the queue according to the policy.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def add_request(self, request: Request) -> None:
    """Add a request to the queue according to the policy."""
    pass

`peek_request()` `abstractmethod` ¶

Peek at the request at the front of the queue without removing it.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def peek_request(self) -> Request:
    """Peek at the request at the front of the queue without removing it."""
    pass

`pop_request()` `abstractmethod` ¶

Pop a request from the queue according to the policy.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def pop_request(self) -> Request:
    """Pop a request from the queue according to the policy."""
    pass

`prepend_request(request)` `abstractmethod` ¶

Prepend a request to the front of the queue.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def prepend_request(self, request: Request) -> None:
    """Prepend a request to the front of the queue."""
    pass

`prepend_requests(requests)` `abstractmethod` ¶

Prepend all requests from another queue to the front of this queue.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def prepend_requests(self, requests: "RequestQueue") -> None:
    """Prepend all requests from another queue to the front of this
    queue."""
    pass

`remove_request(request)` `abstractmethod` ¶

Remove a specific request from the queue.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def remove_request(self, request: Request) -> None:
    """Remove a specific request from the queue."""
    pass

`remove_requests(requests)` `abstractmethod` ¶

Remove multiple specific requests from the queue.

Source code in vllm/v1/core/sched/request_queue.py

@abstractmethod
def remove_requests(self, requests: Iterable[Request]) -> None:
    """Remove multiple specific requests from the queue."""
    pass

`SchedulingPolicy` ¶

Bases: Enum

Enum for scheduling policies.

Source code in vllm/v1/core/sched/request_queue.py

class SchedulingPolicy(Enum):
    """Enum for scheduling policies."""

    FCFS = "fcfs"
    PRIORITY = "priority"

`create_request_queue(policy)` ¶

Create request queue based on scheduling policy.

Source code in vllm/v1/core/sched/request_queue.py

def create_request_queue(policy: SchedulingPolicy) -> RequestQueue:
    """Create request queue based on scheduling policy."""
    if policy == SchedulingPolicy.PRIORITY:
        return PriorityRequestQueue()
    elif policy == SchedulingPolicy.FCFS:
        return FCFSRequestQueue()
    else:
        raise ValueError(f"Unknown scheduling policy: {policy}")

vllm.v1.core.sched.request_queue ¶

FCFSRequestQueue ¶

__bool__() ¶

__iter__() ¶

__len__() ¶

add_request(request) ¶

peek_request() ¶

pop_request() ¶

prepend_request(request) ¶

prepend_requests(requests) ¶

remove_request(request) ¶

remove_requests(requests) ¶

PriorityRequestQueue ¶

__bool__() ¶

__iter__() ¶

__len__() ¶

add_request(request) ¶

peek_request() ¶

pop_request() ¶

prepend_request(request) ¶

prepend_requests(requests) ¶

remove_request(request) ¶

remove_requests(requests) ¶

RequestQueue ¶

__bool__() abstractmethod ¶

__iter__() abstractmethod ¶

__len__() abstractmethod ¶

add_request(request) abstractmethod ¶

peek_request() abstractmethod ¶

pop_request() abstractmethod ¶

prepend_request(request) abstractmethod ¶

prepend_requests(requests) abstractmethod ¶

remove_request(request) abstractmethod ¶

remove_requests(requests) abstractmethod ¶

SchedulingPolicy ¶

create_request_queue(policy) ¶

`vllm.v1.core.sched.request_queue` ¶

`FCFSRequestQueue` ¶

`bool()` ¶

`iter()` ¶

`len()` ¶

`add_request(request)` ¶

`peek_request()` ¶

`pop_request()` ¶

`prepend_request(request)` ¶

`prepend_requests(requests)` ¶

`remove_request(request)` ¶

`remove_requests(requests)` ¶

`PriorityRequestQueue` ¶

`bool()` ¶

`iter()` ¶

`len()` ¶

`add_request(request)` ¶

`peek_request()` ¶

`pop_request()` ¶

`prepend_request(request)` ¶

`prepend_requests(requests)` ¶

`remove_request(request)` ¶

`remove_requests(requests)` ¶

`RequestQueue` ¶

`bool()` `abstractmethod` ¶

`iter()` `abstractmethod` ¶

`len()` `abstractmethod` ¶

`add_request(request)` `abstractmethod` ¶

`peek_request()` `abstractmethod` ¶

`pop_request()` `abstractmethod` ¶

`prepend_request(request)` `abstractmethod` ¶

`prepend_requests(requests)` `abstractmethod` ¶

`remove_request(request)` `abstractmethod` ¶

`remove_requests(requests)` `abstractmethod` ¶

`SchedulingPolicy` ¶

`create_request_queue(policy)` ¶