text-generation-inference/backends/v3/src/main.rs

use clap::{Parser, Subcommand};
use text_generation_router::server;
use text_generation_router_v3::{connect_backend, V3Error};
use thiserror::Error;

/// App Configuration
#[derive(Parser, Debug)]
#[clap(author, version, about, long_about = None)]
struct Args {
    #[command(subcommand)]
    command: Option<Commands>,

    #[clap(default_value = "128", long, env)]
    max_concurrent_requests: usize,
    #[clap(default_value = "2", long, env)]
    max_best_of: usize,
    #[clap(default_value = "4", long, env)]
    max_stop_sequences: usize,
    #[clap(default_value = "5", long, env)]
    max_top_n_tokens: u32,
    #[clap(default_value = "1024", long, env)]
    max_input_tokens: usize,
    #[clap(default_value = "2048", long, env)]
    max_total_tokens: usize,
    #[clap(default_value = "1.2", long, env)]
    waiting_served_ratio: f32,
    #[clap(default_value = "4096", long, env)]
    max_batch_prefill_tokens: u32,
    #[clap(long, env)]
    max_batch_total_tokens: Option<u32>,
    #[clap(default_value = "20", long, env)]
    max_waiting_tokens: usize,
    #[clap(long, env)]
    max_batch_size: Option<usize>,
    #[clap(default_value = "0.0.0.0", long, env)]
    hostname: String,
    #[clap(default_value = "3000", long, short, env)]
    port: u16,
    #[clap(default_value = "/tmp/text-generation-server-0", long, env)]
    master_shard_uds_path: String,
    #[clap(default_value = "bigscience/bloom", long, env)]
    tokenizer_name: String,
    #[clap(long, env)]
    tokenizer_config_path: Option<String>,
    #[clap(long, env)]
    revision: Option<String>,
    #[clap(default_value = "2", long, env)]
    validation_workers: usize,
    #[clap(long, env)]
    api_key: Option<String>,
    #[clap(long, env)]
    json_output: bool,
    #[clap(long, env)]
    otlp_endpoint: Option<String>,
    #[clap(default_value = "text-generation-inference.router", long, env)]
    otlp_service_name: String,
    #[clap(long, env)]
    cors_allow_origin: Option<Vec<String>>,
    #[clap(long, env)]
    ngrok: bool,
    #[clap(long, env)]
    ngrok_authtoken: Option<String>,
    #[clap(long, env)]
    ngrok_edge: Option<String>,
    #[clap(long, env, default_value_t = false)]
    messages_api_enabled: bool,
    #[clap(long, env, default_value_t = false)]
    disable_grammar_support: bool,
    #[clap(default_value = "4", long, env)]
    max_client_batch_size: usize,
    #[clap(long, env, default_value_t)]
    disable_usage_stats: bool,
    #[clap(long, env, default_value_t)]
    disable_crash_reports: bool,
}

#[derive(Debug, Subcommand)]
enum Commands {
    PrintSchema,
}

#[tokio::main]
async fn main() -> Result<(), RouterError> {
    // Get args
    let args = Args::parse();
    // Pattern match configuration
    let Args {
        command,
        max_concurrent_requests,
        max_best_of,
        max_stop_sequences,
        max_top_n_tokens,
        max_input_tokens,
        max_total_tokens,
        waiting_served_ratio,
        max_batch_prefill_tokens,
        max_batch_total_tokens,
        max_waiting_tokens,
        max_batch_size,
        hostname,
        port,
        master_shard_uds_path,
        tokenizer_name,
        tokenizer_config_path,
        revision,
        validation_workers,
        api_key,
        json_output,
        otlp_endpoint,
        otlp_service_name,
        cors_allow_origin,
        ngrok,
        ngrok_authtoken,
        ngrok_edge,
        messages_api_enabled,
        disable_grammar_support,
        disable_usage_stats,
        disable_crash_reports,
        max_client_batch_size,
    } = args;

    if let Some(Commands::PrintSchema) = command {
        use utoipa::OpenApi;
        let api_doc = text_generation_router::server::ApiDoc::openapi();
        let api_doc = serde_json::to_string_pretty(&api_doc).unwrap();
        println!("{}", api_doc);
        std::process::exit(0);
    };
    text_generation_router::logging::init_logging(otlp_endpoint, otlp_service_name, json_output);

    // Validate args
    if max_input_tokens >= max_total_tokens {
        return Err(RouterError::ArgumentValidation(
            "`max_input_tokens` must be < `max_total_tokens`".to_string(),
        ));
    }
    if max_input_tokens as u32 > max_batch_prefill_tokens {
        return Err(RouterError::ArgumentValidation(format!("`max_batch_prefill_tokens` must be >= `max_input_tokens`. Given: {max_batch_prefill_tokens} and {max_input_tokens}")));
    }

    if validation_workers == 0 {
        return Err(RouterError::ArgumentValidation(
            "`validation_workers` must be > 0".to_string(),
        ));
    }

    if let Some(ref max_batch_total_tokens) = max_batch_total_tokens {
        if max_batch_prefill_tokens > *max_batch_total_tokens {
            return Err(RouterError::ArgumentValidation(format!("`max_batch_prefill_tokens` must be <= `max_batch_total_tokens`. Given: {max_batch_prefill_tokens} and {max_batch_total_tokens}")));
        }
        if max_total_tokens as u32 > *max_batch_total_tokens {
            return Err(RouterError::ArgumentValidation(format!("`max_total_tokens` must be <= `max_batch_total_tokens`. Given: {max_total_tokens} and {max_batch_total_tokens}")));
        }
    }

    let (backend, _backend_info) = connect_backend(
        max_input_tokens,
        max_total_tokens,
        master_shard_uds_path,
        waiting_served_ratio,
        max_batch_prefill_tokens,
        max_batch_total_tokens,
        max_waiting_tokens,
        max_batch_size,
    )
    .await?;

    // Run server
    server::run(
        backend,
        max_concurrent_requests,
        max_best_of,
        max_stop_sequences,
        max_top_n_tokens,
        max_input_tokens,
        max_total_tokens,
        validation_workers,
        api_key,
        tokenizer_name,
        tokenizer_config_path,
        revision,
        hostname,
        port,
        cors_allow_origin,
        ngrok,
        ngrok_authtoken,
        ngrok_edge,
        messages_api_enabled,
        disable_grammar_support,
        max_client_batch_size,
        disable_usage_stats,
        disable_crash_reports,
    )
    .await?;
    Ok(())
}

#[derive(Debug, Error)]
enum RouterError {
    #[error("Argument validation error: {0}")]
    ArgumentValidation(String),
    #[error("Backend failed: {0}")]
    Backend(#[from] V3Error),
    #[error("WebServer error: {0}")]
    WebServer(#[from] server::WebServerError),
    #[error("Tokio runtime failed to start: {0}")]
    Tokio(#[from] std::io::Error),
}
Rebase TRT-llm (#2331) * wip wip refacto refacto Initial setup for CXX binding to TRTLLM Working FFI call for TGI and TRTLLM backend Remove unused parameters annd force tokenizer name to be set Overall build TRTLLM and deps through CMake build system Enable end to end CMake build First version loading engines and making it ready for inference Remembering to check how we can detect support for chunked context Move to latest TensorRT-LLM version Specify which default log level to use depending on CMake build type make leader executor mode working unconditionally call InitializeBackend on the FFI layer bind to CUDA::nvml to retrieve compute capabilities at runtime updated logic and comment to detect cuda compute capabilities implement the Stream method to send new tokens through a callback use spdlog release 1.14.1 moving forward update trtllm to latest version a96cccafcf6365c128f004f779160951f8c0801c correctly tell cmake to build dependent tensorrt-llm required libraries create cmake install target to put everything relevant in installation folder add auth_token CLI argument to provide hf hub authentification token allow converting huggingface::tokenizers error to TensorRtLlmBackendError use correct include for spdlog include guard to build example in cmakelists working setup of the ffi layer remove fmt import use external fmt lib end to end ffi flow working make sure to track include/ffi.h to trigger rebuild from cargo impl the rust backend which currently cannot move the actual computation in background thread expose shutdown function at ffi layer impl RwLock scenario for TensorRtLllmBackend oops missing c++ backend definitions compute the number of maximum new tokens for each request independently make sure the context is not dropped in the middle of the async decoding. remove unnecessary log add all the necessary plumbery to return the generated content update invalid doc in cpp file correctly forward back the log probabilities remove unneeded scope variable for now refactor Stream impl for Generation to factorise code expose the internal missing start/queue timestamp forward tgi parameters rep/freq penalty add some more validation about grammar not supported define a shared struct to hold the result of a decoding step expose information about potential error happening while decoding remove logging add logging in case of decoding error make sure executor_worker is provided add initial Dockerfile for TRTLLM backend add some more information in CMakeLists.txt to correctly install executorWorker add some more information in CMakeLists.txt to correctly find and install nvrtc wrapper simplify prebuilt trtllm libraries name definition do the same name definition stuff for tensorrt_llm_executor_static leverage pkg-config to probe libraries paths and reuse new install structure from cmake fix bad copy/past missing nvinfer linkage direction align all the linker search dependency add missing pkgconfig folder for MPI in Dockerfile correctly setup linking search path for runtime layer fix missing / before tgi lib path adding missing ld_library_path for cuda stubs in Dockerfile update tgi entrypoint commenting out Python part for TensorRT installation refactored docker image move to TensorRT-LLM v0.11.0 make docker linter happy with same capitalization rule fix typo refactor the compute capabilities detection along with num gpus update TensorRT-LLM to latest version update TensorRT install script to latest update build.rs to link to cuda 12.5 add missing dependant libraries for linking clean up a bit install to decoder_attention target add some custom stuff for nccl linkage fix envvar CARGO_CFG_TARGET_ARCH set at runtime vs compile time use std::env::const::ARCH make sure variable live long enough... look for cuda 12.5 add some more basic info in README.md * Rebase. * Fix autodocs. * Let's try to enable trtllm backend. * Ignore backends/v3 by default. * Fixing client. * Fix makefile + autodocs. * Updating the schema thing + redocly. * Fix trtllm lint. * Adding pb files ? * Remove cargo fmt temporarily. * ? * Tmp. * Remove both check + clippy ? * Backporting telemetry. * Backporting 457fb0a1 * Remove PB from git. * Fixing PB with default member backends/client * update TensorRT-LLM to latest version * provided None for api_key * link against libtensorrt_llm and not libtensorrt-llm --------- Co-authored-by: OlivierDehaene <23298448+OlivierDehaene@users.noreply.github.com> Co-authored-by: Morgan Funtowicz <morgan@huggingface.co> 2024-07-31 08:33:10 +00:00			`use clap::{Parser, Subcommand};`
			`use text_generation_router::server;`
			`use text_generation_router_v3::{connect_backend, V3Error};`
			`use thiserror::Error;`

			`/// App Configuration`
			`#[derive(Parser, Debug)]`
			`#[clap(author, version, about, long_about = None)]`
			`struct Args {`
			`#[command(subcommand)]`
			`command: Option<Commands>,`

			`#[clap(default_value = "128", long, env)]`
			`max_concurrent_requests: usize,`
			`#[clap(default_value = "2", long, env)]`
			`max_best_of: usize,`
			`#[clap(default_value = "4", long, env)]`
			`max_stop_sequences: usize,`
			`#[clap(default_value = "5", long, env)]`
			`max_top_n_tokens: u32,`
			`#[clap(default_value = "1024", long, env)]`
			`max_input_tokens: usize,`
			`#[clap(default_value = "2048", long, env)]`
			`max_total_tokens: usize,`
			`#[clap(default_value = "1.2", long, env)]`
			`waiting_served_ratio: f32,`
			`#[clap(default_value = "4096", long, env)]`
			`max_batch_prefill_tokens: u32,`
			`#[clap(long, env)]`
			`max_batch_total_tokens: Option<u32>,`
			`#[clap(default_value = "20", long, env)]`
			`max_waiting_tokens: usize,`
			`#[clap(long, env)]`
			`max_batch_size: Option<usize>,`
			`#[clap(default_value = "0.0.0.0", long, env)]`
			`hostname: String,`
			`#[clap(default_value = "3000", long, short, env)]`
			`port: u16,`
			`#[clap(default_value = "/tmp/text-generation-server-0", long, env)]`
			`master_shard_uds_path: String,`
			`#[clap(default_value = "bigscience/bloom", long, env)]`
			`tokenizer_name: String,`
			`#[clap(long, env)]`
			`tokenizer_config_path: Option<String>,`
			`#[clap(long, env)]`
			`revision: Option<String>,`
			`#[clap(default_value = "2", long, env)]`
			`validation_workers: usize,`
			`#[clap(long, env)]`
			`api_key: Option<String>,`
			`#[clap(long, env)]`
			`json_output: bool,`
			`#[clap(long, env)]`
			`otlp_endpoint: Option<String>,`
			`#[clap(default_value = "text-generation-inference.router", long, env)]`
			`otlp_service_name: String,`
			`#[clap(long, env)]`
			`cors_allow_origin: Option<Vec<String>>,`
			`#[clap(long, env)]`
			`ngrok: bool,`
			`#[clap(long, env)]`
			`ngrok_authtoken: Option<String>,`
			`#[clap(long, env)]`
			`ngrok_edge: Option<String>,`
			`#[clap(long, env, default_value_t = false)]`
			`messages_api_enabled: bool,`
			`#[clap(long, env, default_value_t = false)]`
			`disable_grammar_support: bool,`
			`#[clap(default_value = "4", long, env)]`
			`max_client_batch_size: usize,`
			`#[clap(long, env, default_value_t)]`
			`disable_usage_stats: bool,`
			`#[clap(long, env, default_value_t)]`
			`disable_crash_reports: bool,`
			`}`

			`#[derive(Debug, Subcommand)]`
			`enum Commands {`
			`PrintSchema,`
			`}`

			`#[tokio::main]`
			`async fn main() -> Result<(), RouterError> {`
			`// Get args`
			`let args = Args::parse();`
			`// Pattern match configuration`
			`let Args {`
			`command,`
			`max_concurrent_requests,`
			`max_best_of,`
			`max_stop_sequences,`
			`max_top_n_tokens,`
			`max_input_tokens,`
			`max_total_tokens,`
			`waiting_served_ratio,`
			`max_batch_prefill_tokens,`
			`max_batch_total_tokens,`
			`max_waiting_tokens,`
			`max_batch_size,`
			`hostname,`
			`port,`
			`master_shard_uds_path,`
			`tokenizer_name,`
			`tokenizer_config_path,`
			`revision,`
			`validation_workers,`
			`api_key,`
			`json_output,`
			`otlp_endpoint,`
			`otlp_service_name,`
			`cors_allow_origin,`
			`ngrok,`
			`ngrok_authtoken,`
			`ngrok_edge,`
			`messages_api_enabled,`
			`disable_grammar_support,`
			`disable_usage_stats,`
			`disable_crash_reports,`
			`max_client_batch_size,`
			`} = args;`

			`if let Some(Commands::PrintSchema) = command {`
			`use utoipa::OpenApi;`
			`let api_doc = text_generation_router::server::ApiDoc::openapi();`
			`let api_doc = serde_json::to_string_pretty(&api_doc).unwrap();`
			`println!("{}", api_doc);`
			`std::process::exit(0);`
			`};`
			`text_generation_router::logging::init_logging(otlp_endpoint, otlp_service_name, json_output);`

			`// Validate args`
			`if max_input_tokens >= max_total_tokens {`
			`return Err(RouterError::ArgumentValidation(`
			"`max_input_tokens` must be < `max_total_tokens`".to_string(),
			`));`
			`}`
			`if max_input_tokens as u32 > max_batch_prefill_tokens {`
			return Err(RouterError::ArgumentValidation(format!("`max_batch_prefill_tokens` must be >= `max_input_tokens`. Given: {max_batch_prefill_tokens} and {max_input_tokens}")));
			`}`

			`if validation_workers == 0 {`
			`return Err(RouterError::ArgumentValidation(`
			"`validation_workers` must be > 0".to_string(),
			`));`
			`}`

			`if let Some(ref max_batch_total_tokens) = max_batch_total_tokens {`
			`if max_batch_prefill_tokens > *max_batch_total_tokens {`
			return Err(RouterError::ArgumentValidation(format!("`max_batch_prefill_tokens` must be <= `max_batch_total_tokens`. Given: {max_batch_prefill_tokens} and {max_batch_total_tokens}")));
			`}`
			`if max_total_tokens as u32 > *max_batch_total_tokens {`
			return Err(RouterError::ArgumentValidation(format!("`max_total_tokens` must be <= `max_batch_total_tokens`. Given: {max_total_tokens} and {max_batch_total_tokens}")));
			`}`
			`}`

			`let (backend, _backend_info) = connect_backend(`
			`max_input_tokens,`
			`max_total_tokens,`
			`master_shard_uds_path,`
			`waiting_served_ratio,`
			`max_batch_prefill_tokens,`
			`max_batch_total_tokens,`
			`max_waiting_tokens,`
			`max_batch_size,`
			`)`
			`.await?;`

			`// Run server`
			`server::run(`
			`backend,`
			`max_concurrent_requests,`
			`max_best_of,`
			`max_stop_sequences,`
			`max_top_n_tokens,`
			`max_input_tokens,`
			`max_total_tokens,`
			`validation_workers,`
			`api_key,`
			`tokenizer_name,`
			`tokenizer_config_path,`
			`revision,`
			`hostname,`
			`port,`
			`cors_allow_origin,`
			`ngrok,`
			`ngrok_authtoken,`
			`ngrok_edge,`
			`messages_api_enabled,`
			`disable_grammar_support,`
			`max_client_batch_size,`
			`disable_usage_stats,`
			`disable_crash_reports,`
			`)`
			`.await?;`
			`Ok(())`
			`}`

			`#[derive(Debug, Error)]`
			`enum RouterError {`
			`#[error("Argument validation error: {0}")]`
			`ArgumentValidation(String),`
			`#[error("Backend failed: {0}")]`
			`Backend(#[from] V3Error),`
			`#[error("WebServer error: {0}")]`
			`WebServer(#[from] server::WebServerError),`
			`#[error("Tokio runtime failed to start: {0}")]`
			`Tokio(#[from] std::io::Error),`
			`}`