Skip to content

vllm.entrypoints.scale_out.token_in_token_out.api_router

Functions:

abort_requests(raw_request) async

Abort one or more requests. To be used in a Disaggregated Everything setup.

Source code in vllm/entrypoints/scale_out/token_in_token_out/api_router.py
async def abort_requests(raw_request: Request):
    """Abort one or more requests. To be used in a
    Disaggregated Everything setup.
    """
    try:
        body = await raw_request.json()
    except json.JSONDecodeError as e:
        raise HTTPException(
            status_code=HTTPStatus.BAD_REQUEST.value,
            detail=f"JSON decode error: {e}",
        ) from e
    request_ids = body.get("request_ids")
    if request_ids is None:
        raise HTTPException(
            status_code=HTTPStatus.BAD_REQUEST.value,
            detail="Missing 'request_ids' in request body",
        )
    # Abort requests in background
    asyncio.create_task(engine_client(raw_request).abort(request_ids))
    return Response(status_code=200)