Skip to content

vllm.entrypoints.serve.lora.api_router

Functions:

  • attach_router –

    Attach the LoRA adapter load/unload endpoints to the API server.

_attach_router(app)

Register the LoRA adapter load/unload routes on the app.

Source code in vllm/entrypoints/serve/lora/api_router.py
def _attach_router(app: FastAPI):
    """Register the LoRA adapter load/unload routes on the app."""
    import model_hosting_container_standards.sagemaker as sagemaker_standards

    @sagemaker_standards.register_load_adapter_handler(
        request_shape={
            "lora_name": "body.name",
            "lora_path": "body.src",
            "load_inplace": "body.load_inplace || `false`",
            "is_3d_lora_weight": "body.is_3d_lora_weight || `false`",
        },
    )
    @router.post("/v1/load_lora_adapter", dependencies=[Depends(validate_json_request)])
    async def load_lora_adapter(request: LoadLoRAAdapterRequest, raw_request: Request):
        """Handle POST /v1/load_lora_adapter: load a LoRA adapter into the
        serving engine."""
        handler: OpenAIServingModels = models(raw_request)
        response = await handler.load_lora_adapter(request)
        if isinstance(response, ErrorResponse):
            return JSONResponse(
                content=response.model_dump(), status_code=response.error.code
            )

        return Response(status_code=200, content=response)

    @sagemaker_standards.register_unload_adapter_handler(
        request_shape={
            "lora_name": "path_params.adapter_name",
        }
    )
    @router.post(
        "/v1/unload_lora_adapter", dependencies=[Depends(validate_json_request)]
    )
    async def unload_lora_adapter(
        request: UnloadLoRAAdapterRequest, raw_request: Request
    ):
        """Handle POST /v1/unload_lora_adapter: unload a LoRA adapter from
        the serving engine."""
        handler: OpenAIServingModels = models(raw_request)
        response = await handler.unload_lora_adapter(request)
        if isinstance(response, ErrorResponse):
            return JSONResponse(
                content=response.model_dump(), status_code=response.error.code
            )

        return Response(status_code=200, content=response)

    # register the router
    app.include_router(router)

attach_router(app)

Attach the LoRA adapter load/unload endpoints to the API server.

Does nothing when dynamic LoRA updating is disabled. Handler levels are snapshotted and restored because importing model_hosting_container_standards may reconfigure root logging.

Source code in vllm/entrypoints/serve/lora/api_router.py
def attach_router(app: FastAPI):
    """Attach the LoRA adapter load/unload endpoints to the API server.

    Does nothing when dynamic LoRA updating is disabled. Handler levels are
    snapshotted and restored because importing
    model_hosting_container_standards may reconfigure root logging.
    """
    if not envs.VLLM_ALLOW_RUNTIME_LORA_UPDATING:
        """If LoRA dynamic loading & unloading is not enabled, do nothing."""
        return
    logger.warning(
        "LoRA dynamic loading & unloading is enabled in the API server. "
        "This should ONLY be used for local development!"
    )

    snapshot = [
        (h, h.level)
        for lg in (logging.getLogger(), logging.getLogger("vllm"))
        for h in lg.handlers
    ]

    try:
        _attach_router(app)
    finally:
        for handler, level in snapshot:
            handler.setLevel(level)